Re: [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue for gfx11

Eric Huang <[email protected]>
Newsgroups org.freedesktop.lists.amd-gfx
Message-ID <[email protected]>
Ping...

On 2026-08-05 17:27, Eric Huang wrote:
> the issue only happens with oversubscription when gpu has no
> workload, the root cause is mes oversubscription timer, so
> disable mes timer and make a similar timer in kfd to resolve
> the issue.
>
> Signed-off-by: Eric Huang <[email protected]>
> ---
>   drivers/gpu/drm/amd/amdgpu/mes_v11_0.c        |  1 -
>   .../drm/amd/amdkfd/kfd_device_queue_manager.c | 39 ++++++++++++++++++-
>   .../drm/amd/amdkfd/kfd_device_queue_manager.h |  1 +
>   3 files changed, 39 insertions(+), 2 deletions(-)
>
> diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> index bc0ee6ccfca7..40683db66a13 100644
> --- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> +++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> @@ -1019,7 +1019,6 @@ static int mes_v11_0_set_hw_resources(struct amdgpu_mes *mes)
>   	mes_set_hw_res_pkt.use_different_vmid_compute = 1;
>   	mes_set_hw_res_pkt.enable_reg_active_poll = 1;
>   	mes_set_hw_res_pkt.enable_level_process_quantum_check = 1;
> -	mes_set_hw_res_pkt.oversubscription_timer = 50;
>   	if (adev->mes.use_rs64mem)
>   		mes_set_hw_res_pkt.use_rs64mem_for_proc_gang_ctx = 1;
>   
> diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> index ea9d87450eae..350e0ce308d3 100644
> --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> @@ -47,6 +47,9 @@
>   /* See unmap_queues_cpsch() */
>   #define USE_DEFAULT_GRACE_PERIOD 0xffffffff
>   
> +/* Interval for notifying MES of work on unmapped queues during oversubscription */
> +#define DQM_MES_UNMAP_NOTIFY_DELAY_MS 50
> +
>   static int set_pasid_vmid_mapping(struct device_queue_manager *dqm,
>   				  u32 pasid, unsigned int vmid);
>   
> @@ -276,8 +279,16 @@ static int add_queue_mes(struct device_queue_manager *dqm, struct queue *q,
>   			q->properties.doorbell_off);
>   		dev_err(adev->dev, "MES might be in unrecoverable state, issue a GPU reset\n");
>   		kfd_hws_hang(dqm);
> +		return r;
>   	}
>   
> +	/* GFX11: start notify timer only once when oversubscription begins */
> +	if (KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
> +	    KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
> +	    dqm->active_cp_queue_count > get_cp_queues_num(dqm))
> +		queue_delayed_work(system_wq, &dqm->notify_unmap_work,
> +				   msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
> +
>   	return r;
>   }
>   
> @@ -354,7 +365,16 @@ static void set_perfcount(struct device_queue_manager *dqm, int enable)
>   static int remove_queue_mes(struct device_queue_manager *dqm, struct queue *q,
>   			    struct qcm_process_device *qpd)
>   {
> -	return remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
> +	int r = remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
> +
> +	/* GFX11: stop notify timer when oversubscription clears */
> +	if (!r &&
> +	    KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
> +	    KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
> +	    dqm->active_cp_queue_count <= get_cp_queues_num(dqm))
> +		cancel_delayed_work(&dqm->notify_unmap_work);
> +
> +	return r;
>   }
>   
>   static int remove_all_kfd_queues_mes(struct device_queue_manager *dqm)
> @@ -3088,6 +3108,20 @@ static void deallocate_hiq_sdma_mqd(struct kfd_node *dev,
>   	amdgpu_amdkfd_free_kernel_mem(dev->adev, &mqd->mem);
>   }
>   
> +static void mes_notify_unmap_work_handler(struct work_struct *work)
> +{
> +	struct device_queue_manager *dqm =
> +		container_of(work, struct device_queue_manager,
> +			     notify_unmap_work.work);
> +
> +	amdgpu_mes_notify_unmap_queue((struct amdgpu_device *)dqm->dev->adev);
> +
> +	/* Re-arm if still oversubscribed */
> +	if (READ_ONCE(dqm->active_cp_queue_count) > get_cp_queues_num(dqm))
> +		queue_delayed_work(system_wq, &dqm->notify_unmap_work,
> +				   msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
> +}
> +
>   struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
>   {
>   	struct device_queue_manager *dqm;
> @@ -3213,6 +3247,8 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
>   
>   	if (!dqm->ops.initialize(dqm)) {
>   		init_waitqueue_head(&dqm->destroy_wait);
> +		INIT_DELAYED_WORK(&dqm->notify_unmap_work,
> +				  mes_notify_unmap_work_handler);
>   		return dqm;
>   	}
>   
> @@ -3229,6 +3265,7 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
>   
>   void device_queue_manager_uninit(struct device_queue_manager *dqm)
>   {
> +	cancel_delayed_work_sync(&dqm->notify_unmap_work);
>   	dqm->ops.stop(dqm);
>   	dqm->ops.uninitialize(dqm);
>   	if (!dqm->dev->kfd->shared_resources.enable_mes)
> diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> index c9f9f7a87111..21cf3c16f3f9 100644
> --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> @@ -281,6 +281,7 @@ struct device_queue_manager {
>   	uint32_t		wait_times;
>   
>   	wait_queue_head_t	destroy_wait;
> +	struct delayed_work	notify_unmap_work;
>   
>   	/* for per-queue reset support */
>   	struct dqm_detect_hang_info *detect_hang_info;
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.