All of lore.kernel.org
 help / color / mirror / Atom feed
* [PATCH 1/2] drm/amdgpu: add new mes misc api for gfx11
@ 2026-08-05 21:27 Eric Huang
  2026-08-05 21:27 ` [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue " Eric Huang
  2026-08-13 15:34 ` [PATCH 1/2] drm/amdgpu: add new mes misc api " Alex Deucher
  0 siblings, 2 replies; 5+ messages in thread
From: Eric Huang @ 2026-08-05 21:27 UTC (permalink / raw)
  To: amd-gfx; +Cc: Tishko.Araz, Eric Huang

new api will be used to workaround a HW scheduler issue
for 100% usage when gpu has no workload.

Signed-off-by: Eric Huang <jinhuieric.huang@amd.com>
---
 drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c | 23 +++++++++++++++++++++++
 drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h |  4 ++++
 drivers/gpu/drm/amd/amdgpu/mes_v11_0.c  |  5 ++++-
 3 files changed, 31 insertions(+), 1 deletion(-)

diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c
index b96f94e5169f..421d937c1188 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c
@@ -1140,6 +1140,29 @@ void amdgpu_mes_free_gang_ctx_index(struct amdgpu_mes *mes,
 	amdgpu_mes_unlock(mes);
 }
 
+int amdgpu_mes_notify_unmap_queue(struct amdgpu_device *adev)
+{
+	struct mes_misc_op_input op_input = {0};
+	int r;
+
+	op_input.op = MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE;
+
+	if (!adev->mes.funcs->misc_op) {
+		dev_err(adev->dev, "mes notify unmap queue is not supported!\n");
+		r = -EINVAL;
+		goto error;
+	}
+
+	amdgpu_mes_lock(&adev->mes);
+	r = adev->mes.funcs->misc_op(&adev->mes, &op_input);
+	amdgpu_mes_unlock(&adev->mes);
+	if (r)
+		dev_err(adev->dev, "failed to notify unmap queue.\n");
+
+error:
+	return r;
+}
+
 #if defined(CONFIG_DEBUG_FS)
 
 static int amdgpu_debugfs_mes_event_log_show(struct seq_file *m, void *unused)
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h
index 977c057dcce8..5230b1ba1a46 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h
@@ -366,6 +366,7 @@ enum mes_misc_opcode {
 	MES_MISC_OP_WRM_REG_WR_WAIT,
 	MES_MISC_OP_SET_SHADER_DEBUGGER,
 	MES_MISC_OP_CHANGE_CONFIG,
+	MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE
 };
 
 struct mes_misc_op_input {
@@ -643,4 +644,7 @@ int amdgpu_mes_alloc_gang_ctx_index(struct amdgpu_mes *mes,
 				    uint32_t *index);
 void amdgpu_mes_free_gang_ctx_index(struct amdgpu_mes *mes,
 				    uint32_t index);
+
+int amdgpu_mes_notify_unmap_queue(struct amdgpu_device *adev);
+
 #endif /* __AMDGPU_MES_H__ */
diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
index a0b874c9ee7b..bc0ee6ccfca7 100644
--- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
@@ -957,7 +957,10 @@ static int mes_v11_0_misc_op(struct amdgpu_mes *mes,
 		misc_pkt.change_config.option.bits.limit_single_process =
 				input->change_config.option.limit_single_process;
 		break;
-
+	case MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE:
+		misc_pkt.opcode = MESAPI_MISC__NOTIFY_WORK_ON_UNMAPPED_QUEUE;
+		misc_pkt.queue_sch_level = AMD_PRIORITY_LEVEL_NORMAL;
+		break;
 	default:
 		drm_err(adev_to_drm(mes->adev), "unsupported misc op (%d)\n", input->op);
 		return -EINVAL;
-- 
2.34.1


^ permalink raw reply related	[flat|nested] 5+ messages in thread

* [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue for gfx11
  2026-08-05 21:27 [PATCH 1/2] drm/amdgpu: add new mes misc api for gfx11 Eric Huang
@ 2026-08-05 21:27 ` Eric Huang
  2026-08-13 15:04   ` Eric Huang
  2026-08-13 15:35   ` Alex Deucher
  2026-08-13 15:34 ` [PATCH 1/2] drm/amdgpu: add new mes misc api " Alex Deucher
  1 sibling, 2 replies; 5+ messages in thread
From: Eric Huang @ 2026-08-05 21:27 UTC (permalink / raw)
  To: amd-gfx; +Cc: Tishko.Araz, Eric Huang

the issue only happens with oversubscription when gpu has no
workload, the root cause is mes oversubscription timer, so
disable mes timer and make a similar timer in kfd to resolve
the issue.

Signed-off-by: Eric Huang <jinhuieric.huang@amd.com>
---
 drivers/gpu/drm/amd/amdgpu/mes_v11_0.c        |  1 -
 .../drm/amd/amdkfd/kfd_device_queue_manager.c | 39 ++++++++++++++++++-
 .../drm/amd/amdkfd/kfd_device_queue_manager.h |  1 +
 3 files changed, 39 insertions(+), 2 deletions(-)

diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
index bc0ee6ccfca7..40683db66a13 100644
--- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
@@ -1019,7 +1019,6 @@ static int mes_v11_0_set_hw_resources(struct amdgpu_mes *mes)
 	mes_set_hw_res_pkt.use_different_vmid_compute = 1;
 	mes_set_hw_res_pkt.enable_reg_active_poll = 1;
 	mes_set_hw_res_pkt.enable_level_process_quantum_check = 1;
-	mes_set_hw_res_pkt.oversubscription_timer = 50;
 	if (adev->mes.use_rs64mem)
 		mes_set_hw_res_pkt.use_rs64mem_for_proc_gang_ctx = 1;
 
diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
index ea9d87450eae..350e0ce308d3 100644
--- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
+++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
@@ -47,6 +47,9 @@
 /* See unmap_queues_cpsch() */
 #define USE_DEFAULT_GRACE_PERIOD 0xffffffff
 
+/* Interval for notifying MES of work on unmapped queues during oversubscription */
+#define DQM_MES_UNMAP_NOTIFY_DELAY_MS 50
+
 static int set_pasid_vmid_mapping(struct device_queue_manager *dqm,
 				  u32 pasid, unsigned int vmid);
 
@@ -276,8 +279,16 @@ static int add_queue_mes(struct device_queue_manager *dqm, struct queue *q,
 			q->properties.doorbell_off);
 		dev_err(adev->dev, "MES might be in unrecoverable state, issue a GPU reset\n");
 		kfd_hws_hang(dqm);
+		return r;
 	}
 
+	/* GFX11: start notify timer only once when oversubscription begins */
+	if (KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
+	    KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
+	    dqm->active_cp_queue_count > get_cp_queues_num(dqm))
+		queue_delayed_work(system_wq, &dqm->notify_unmap_work,
+				   msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
+
 	return r;
 }
 
@@ -354,7 +365,16 @@ static void set_perfcount(struct device_queue_manager *dqm, int enable)
 static int remove_queue_mes(struct device_queue_manager *dqm, struct queue *q,
 			    struct qcm_process_device *qpd)
 {
-	return remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
+	int r = remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
+
+	/* GFX11: stop notify timer when oversubscription clears */
+	if (!r &&
+	    KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
+	    KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
+	    dqm->active_cp_queue_count <= get_cp_queues_num(dqm))
+		cancel_delayed_work(&dqm->notify_unmap_work);
+
+	return r;
 }
 
 static int remove_all_kfd_queues_mes(struct device_queue_manager *dqm)
@@ -3088,6 +3108,20 @@ static void deallocate_hiq_sdma_mqd(struct kfd_node *dev,
 	amdgpu_amdkfd_free_kernel_mem(dev->adev, &mqd->mem);
 }
 
+static void mes_notify_unmap_work_handler(struct work_struct *work)
+{
+	struct device_queue_manager *dqm =
+		container_of(work, struct device_queue_manager,
+			     notify_unmap_work.work);
+
+	amdgpu_mes_notify_unmap_queue((struct amdgpu_device *)dqm->dev->adev);
+
+	/* Re-arm if still oversubscribed */
+	if (READ_ONCE(dqm->active_cp_queue_count) > get_cp_queues_num(dqm))
+		queue_delayed_work(system_wq, &dqm->notify_unmap_work,
+				   msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
+}
+
 struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
 {
 	struct device_queue_manager *dqm;
@@ -3213,6 +3247,8 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
 
 	if (!dqm->ops.initialize(dqm)) {
 		init_waitqueue_head(&dqm->destroy_wait);
+		INIT_DELAYED_WORK(&dqm->notify_unmap_work,
+				  mes_notify_unmap_work_handler);
 		return dqm;
 	}
 
@@ -3229,6 +3265,7 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
 
 void device_queue_manager_uninit(struct device_queue_manager *dqm)
 {
+	cancel_delayed_work_sync(&dqm->notify_unmap_work);
 	dqm->ops.stop(dqm);
 	dqm->ops.uninitialize(dqm);
 	if (!dqm->dev->kfd->shared_resources.enable_mes)
diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
index c9f9f7a87111..21cf3c16f3f9 100644
--- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
+++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
@@ -281,6 +281,7 @@ struct device_queue_manager {
 	uint32_t		wait_times;
 
 	wait_queue_head_t	destroy_wait;
+	struct delayed_work	notify_unmap_work;
 
 	/* for per-queue reset support */
 	struct dqm_detect_hang_info *detect_hang_info;
-- 
2.34.1


^ permalink raw reply related	[flat|nested] 5+ messages in thread

* Re: [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue for gfx11
  2026-08-05 21:27 ` [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue " Eric Huang
@ 2026-08-13 15:04   ` Eric Huang
  2026-08-13 15:35   ` Alex Deucher
  1 sibling, 0 replies; 5+ messages in thread
From: Eric Huang @ 2026-08-13 15:04 UTC (permalink / raw)
  To: amd-gfx

Ping...

On 2026-08-05 17:27, Eric Huang wrote:
> the issue only happens with oversubscription when gpu has no
> workload, the root cause is mes oversubscription timer, so
> disable mes timer and make a similar timer in kfd to resolve
> the issue.
>
> Signed-off-by: Eric Huang <jinhuieric.huang@amd.com>
> ---
>   drivers/gpu/drm/amd/amdgpu/mes_v11_0.c        |  1 -
>   .../drm/amd/amdkfd/kfd_device_queue_manager.c | 39 ++++++++++++++++++-
>   .../drm/amd/amdkfd/kfd_device_queue_manager.h |  1 +
>   3 files changed, 39 insertions(+), 2 deletions(-)
>
> diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> index bc0ee6ccfca7..40683db66a13 100644
> --- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> +++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> @@ -1019,7 +1019,6 @@ static int mes_v11_0_set_hw_resources(struct amdgpu_mes *mes)
>   	mes_set_hw_res_pkt.use_different_vmid_compute = 1;
>   	mes_set_hw_res_pkt.enable_reg_active_poll = 1;
>   	mes_set_hw_res_pkt.enable_level_process_quantum_check = 1;
> -	mes_set_hw_res_pkt.oversubscription_timer = 50;
>   	if (adev->mes.use_rs64mem)
>   		mes_set_hw_res_pkt.use_rs64mem_for_proc_gang_ctx = 1;
>   
> diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> index ea9d87450eae..350e0ce308d3 100644
> --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> @@ -47,6 +47,9 @@
>   /* See unmap_queues_cpsch() */
>   #define USE_DEFAULT_GRACE_PERIOD 0xffffffff
>   
> +/* Interval for notifying MES of work on unmapped queues during oversubscription */
> +#define DQM_MES_UNMAP_NOTIFY_DELAY_MS 50
> +
>   static int set_pasid_vmid_mapping(struct device_queue_manager *dqm,
>   				  u32 pasid, unsigned int vmid);
>   
> @@ -276,8 +279,16 @@ static int add_queue_mes(struct device_queue_manager *dqm, struct queue *q,
>   			q->properties.doorbell_off);
>   		dev_err(adev->dev, "MES might be in unrecoverable state, issue a GPU reset\n");
>   		kfd_hws_hang(dqm);
> +		return r;
>   	}
>   
> +	/* GFX11: start notify timer only once when oversubscription begins */
> +	if (KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
> +	    KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
> +	    dqm->active_cp_queue_count > get_cp_queues_num(dqm))
> +		queue_delayed_work(system_wq, &dqm->notify_unmap_work,
> +				   msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
> +
>   	return r;
>   }
>   
> @@ -354,7 +365,16 @@ static void set_perfcount(struct device_queue_manager *dqm, int enable)
>   static int remove_queue_mes(struct device_queue_manager *dqm, struct queue *q,
>   			    struct qcm_process_device *qpd)
>   {
> -	return remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
> +	int r = remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
> +
> +	/* GFX11: stop notify timer when oversubscription clears */
> +	if (!r &&
> +	    KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
> +	    KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
> +	    dqm->active_cp_queue_count <= get_cp_queues_num(dqm))
> +		cancel_delayed_work(&dqm->notify_unmap_work);
> +
> +	return r;
>   }
>   
>   static int remove_all_kfd_queues_mes(struct device_queue_manager *dqm)
> @@ -3088,6 +3108,20 @@ static void deallocate_hiq_sdma_mqd(struct kfd_node *dev,
>   	amdgpu_amdkfd_free_kernel_mem(dev->adev, &mqd->mem);
>   }
>   
> +static void mes_notify_unmap_work_handler(struct work_struct *work)
> +{
> +	struct device_queue_manager *dqm =
> +		container_of(work, struct device_queue_manager,
> +			     notify_unmap_work.work);
> +
> +	amdgpu_mes_notify_unmap_queue((struct amdgpu_device *)dqm->dev->adev);
> +
> +	/* Re-arm if still oversubscribed */
> +	if (READ_ONCE(dqm->active_cp_queue_count) > get_cp_queues_num(dqm))
> +		queue_delayed_work(system_wq, &dqm->notify_unmap_work,
> +				   msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
> +}
> +
>   struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
>   {
>   	struct device_queue_manager *dqm;
> @@ -3213,6 +3247,8 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
>   
>   	if (!dqm->ops.initialize(dqm)) {
>   		init_waitqueue_head(&dqm->destroy_wait);
> +		INIT_DELAYED_WORK(&dqm->notify_unmap_work,
> +				  mes_notify_unmap_work_handler);
>   		return dqm;
>   	}
>   
> @@ -3229,6 +3265,7 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
>   
>   void device_queue_manager_uninit(struct device_queue_manager *dqm)
>   {
> +	cancel_delayed_work_sync(&dqm->notify_unmap_work);
>   	dqm->ops.stop(dqm);
>   	dqm->ops.uninitialize(dqm);
>   	if (!dqm->dev->kfd->shared_resources.enable_mes)
> diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> index c9f9f7a87111..21cf3c16f3f9 100644
> --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> @@ -281,6 +281,7 @@ struct device_queue_manager {
>   	uint32_t		wait_times;
>   
>   	wait_queue_head_t	destroy_wait;
> +	struct delayed_work	notify_unmap_work;
>   
>   	/* for per-queue reset support */
>   	struct dqm_detect_hang_info *detect_hang_info;


^ permalink raw reply	[flat|nested] 5+ messages in thread

* Re: [PATCH 1/2] drm/amdgpu: add new mes misc api for gfx11
  2026-08-05 21:27 [PATCH 1/2] drm/amdgpu: add new mes misc api for gfx11 Eric Huang
  2026-08-05 21:27 ` [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue " Eric Huang
@ 2026-08-13 15:34 ` Alex Deucher
  1 sibling, 0 replies; 5+ messages in thread
From: Alex Deucher @ 2026-08-13 15:34 UTC (permalink / raw)
  To: Eric Huang; +Cc: amd-gfx, Tishko.Araz

On Wed, Aug 5, 2026 at 6:23 PM Eric Huang <jinhuieric.huang@amd.com> wrote:
>
> new api will be used to workaround a HW scheduler issue
> for 100% usage when gpu has no workload.
>
> Signed-off-by: Eric Huang <jinhuieric.huang@amd.com>
> ---
>  drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c | 23 +++++++++++++++++++++++
>  drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h |  4 ++++
>  drivers/gpu/drm/amd/amdgpu/mes_v11_0.c  |  5 ++++-
>  3 files changed, 31 insertions(+), 1 deletion(-)
>
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c
> index b96f94e5169f..421d937c1188 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c
> @@ -1140,6 +1140,29 @@ void amdgpu_mes_free_gang_ctx_index(struct amdgpu_mes *mes,
>         amdgpu_mes_unlock(mes);
>  }
>
> +int amdgpu_mes_notify_unmap_queue(struct amdgpu_device *adev)
> +{
> +       struct mes_misc_op_input op_input = {0};
> +       int r;
> +
> +       op_input.op = MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE;
> +
> +       if (!adev->mes.funcs->misc_op) {
> +               dev_err(adev->dev, "mes notify unmap queue is not supported!\n");
> +               r = -EINVAL;
> +               goto error;
> +       }
> +
> +       amdgpu_mes_lock(&adev->mes);
> +       r = adev->mes.funcs->misc_op(&adev->mes, &op_input);
> +       amdgpu_mes_unlock(&adev->mes);
> +       if (r)
> +               dev_err(adev->dev, "failed to notify unmap queue.\n");
> +
> +error:
> +       return r;
> +}
> +
>  #if defined(CONFIG_DEBUG_FS)
>
>  static int amdgpu_debugfs_mes_event_log_show(struct seq_file *m, void *unused)
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h
> index 977c057dcce8..5230b1ba1a46 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h
> @@ -366,6 +366,7 @@ enum mes_misc_opcode {
>         MES_MISC_OP_WRM_REG_WR_WAIT,
>         MES_MISC_OP_SET_SHADER_DEBUGGER,
>         MES_MISC_OP_CHANGE_CONFIG,
> +       MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE

What's the minimum fw version which supports this?  We need to check
so we only use it when supported.

Alex

>  };
>
>  struct mes_misc_op_input {
> @@ -643,4 +644,7 @@ int amdgpu_mes_alloc_gang_ctx_index(struct amdgpu_mes *mes,
>                                     uint32_t *index);
>  void amdgpu_mes_free_gang_ctx_index(struct amdgpu_mes *mes,
>                                     uint32_t index);
> +
> +int amdgpu_mes_notify_unmap_queue(struct amdgpu_device *adev);
> +
>  #endif /* __AMDGPU_MES_H__ */
> diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> index a0b874c9ee7b..bc0ee6ccfca7 100644
> --- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> +++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> @@ -957,7 +957,10 @@ static int mes_v11_0_misc_op(struct amdgpu_mes *mes,
>                 misc_pkt.change_config.option.bits.limit_single_process =
>                                 input->change_config.option.limit_single_process;
>                 break;
> -
> +       case MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE:
> +               misc_pkt.opcode = MESAPI_MISC__NOTIFY_WORK_ON_UNMAPPED_QUEUE;
> +               misc_pkt.queue_sch_level = AMD_PRIORITY_LEVEL_NORMAL;
> +               break;
>         default:
>                 drm_err(adev_to_drm(mes->adev), "unsupported misc op (%d)\n", input->op);
>                 return -EINVAL;
> --
> 2.34.1
>

^ permalink raw reply	[flat|nested] 5+ messages in thread

* Re: [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue for gfx11
  2026-08-05 21:27 ` [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue " Eric Huang
  2026-08-13 15:04   ` Eric Huang
@ 2026-08-13 15:35   ` Alex Deucher
  1 sibling, 0 replies; 5+ messages in thread
From: Alex Deucher @ 2026-08-13 15:35 UTC (permalink / raw)
  To: Eric Huang; +Cc: amd-gfx, Tishko.Araz

On Wed, Aug 5, 2026 at 6:23 PM Eric Huang <jinhuieric.huang@amd.com> wrote:
>
> the issue only happens with oversubscription when gpu has no
> workload, the root cause is mes oversubscription timer, so
> disable mes timer and make a similar timer in kfd to resolve
> the issue.

This should only be enabled on FW versions which support it.
Additionally, we need similar treatment for KGD userqs if we make this
change.

Alex

>
> Signed-off-by: Eric Huang <jinhuieric.huang@amd.com>
> ---
>  drivers/gpu/drm/amd/amdgpu/mes_v11_0.c        |  1 -
>  .../drm/amd/amdkfd/kfd_device_queue_manager.c | 39 ++++++++++++++++++-
>  .../drm/amd/amdkfd/kfd_device_queue_manager.h |  1 +
>  3 files changed, 39 insertions(+), 2 deletions(-)
>
> diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> index bc0ee6ccfca7..40683db66a13 100644
> --- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> +++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> @@ -1019,7 +1019,6 @@ static int mes_v11_0_set_hw_resources(struct amdgpu_mes *mes)
>         mes_set_hw_res_pkt.use_different_vmid_compute = 1;
>         mes_set_hw_res_pkt.enable_reg_active_poll = 1;
>         mes_set_hw_res_pkt.enable_level_process_quantum_check = 1;
> -       mes_set_hw_res_pkt.oversubscription_timer = 50;
>         if (adev->mes.use_rs64mem)
>                 mes_set_hw_res_pkt.use_rs64mem_for_proc_gang_ctx = 1;
>
> diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> index ea9d87450eae..350e0ce308d3 100644
> --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> @@ -47,6 +47,9 @@
>  /* See unmap_queues_cpsch() */
>  #define USE_DEFAULT_GRACE_PERIOD 0xffffffff
>
> +/* Interval for notifying MES of work on unmapped queues during oversubscription */
> +#define DQM_MES_UNMAP_NOTIFY_DELAY_MS 50
> +
>  static int set_pasid_vmid_mapping(struct device_queue_manager *dqm,
>                                   u32 pasid, unsigned int vmid);
>
> @@ -276,8 +279,16 @@ static int add_queue_mes(struct device_queue_manager *dqm, struct queue *q,
>                         q->properties.doorbell_off);
>                 dev_err(adev->dev, "MES might be in unrecoverable state, issue a GPU reset\n");
>                 kfd_hws_hang(dqm);
> +               return r;
>         }
>
> +       /* GFX11: start notify timer only once when oversubscription begins */
> +       if (KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
> +           KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
> +           dqm->active_cp_queue_count > get_cp_queues_num(dqm))
> +               queue_delayed_work(system_wq, &dqm->notify_unmap_work,
> +                                  msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
> +
>         return r;
>  }
>
> @@ -354,7 +365,16 @@ static void set_perfcount(struct device_queue_manager *dqm, int enable)
>  static int remove_queue_mes(struct device_queue_manager *dqm, struct queue *q,
>                             struct qcm_process_device *qpd)
>  {
> -       return remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
> +       int r = remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
> +
> +       /* GFX11: stop notify timer when oversubscription clears */
> +       if (!r &&
> +           KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
> +           KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
> +           dqm->active_cp_queue_count <= get_cp_queues_num(dqm))
> +               cancel_delayed_work(&dqm->notify_unmap_work);
> +
> +       return r;
>  }
>
>  static int remove_all_kfd_queues_mes(struct device_queue_manager *dqm)
> @@ -3088,6 +3108,20 @@ static void deallocate_hiq_sdma_mqd(struct kfd_node *dev,
>         amdgpu_amdkfd_free_kernel_mem(dev->adev, &mqd->mem);
>  }
>
> +static void mes_notify_unmap_work_handler(struct work_struct *work)
> +{
> +       struct device_queue_manager *dqm =
> +               container_of(work, struct device_queue_manager,
> +                            notify_unmap_work.work);
> +
> +       amdgpu_mes_notify_unmap_queue((struct amdgpu_device *)dqm->dev->adev);
> +
> +       /* Re-arm if still oversubscribed */
> +       if (READ_ONCE(dqm->active_cp_queue_count) > get_cp_queues_num(dqm))
> +               queue_delayed_work(system_wq, &dqm->notify_unmap_work,
> +                                  msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
> +}
> +
>  struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
>  {
>         struct device_queue_manager *dqm;
> @@ -3213,6 +3247,8 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
>
>         if (!dqm->ops.initialize(dqm)) {
>                 init_waitqueue_head(&dqm->destroy_wait);
> +               INIT_DELAYED_WORK(&dqm->notify_unmap_work,
> +                                 mes_notify_unmap_work_handler);
>                 return dqm;
>         }
>
> @@ -3229,6 +3265,7 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
>
>  void device_queue_manager_uninit(struct device_queue_manager *dqm)
>  {
> +       cancel_delayed_work_sync(&dqm->notify_unmap_work);
>         dqm->ops.stop(dqm);
>         dqm->ops.uninitialize(dqm);
>         if (!dqm->dev->kfd->shared_resources.enable_mes)
> diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> index c9f9f7a87111..21cf3c16f3f9 100644
> --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> @@ -281,6 +281,7 @@ struct device_queue_manager {
>         uint32_t                wait_times;
>
>         wait_queue_head_t       destroy_wait;
> +       struct delayed_work     notify_unmap_work;
>
>         /* for per-queue reset support */
>         struct dqm_detect_hang_info *detect_hang_info;
> --
> 2.34.1
>

^ permalink raw reply	[flat|nested] 5+ messages in thread

end of thread, other threads:[~2026-08-13 15:35 UTC | newest]

Thread overview: 5+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-08-05 21:27 [PATCH 1/2] drm/amdgpu: add new mes misc api for gfx11 Eric Huang
2026-08-05 21:27 ` [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue " Eric Huang
2026-08-13 15:04   ` Eric Huang
2026-08-13 15:35   ` Alex Deucher
2026-08-13 15:34 ` [PATCH 1/2] drm/amdgpu: add new mes misc api " Alex Deucher

This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.