* [PATCH 1/2] drm/amdgpu: add new mes misc api for gfx11
@ 2026-08-05 21:27 Eric Huang
2026-08-05 21:27 ` [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue " Eric Huang
2026-08-13 15:34 ` [PATCH 1/2] drm/amdgpu: add new mes misc api " Alex Deucher
0 siblings, 2 replies; 5+ messages in thread
From: Eric Huang @ 2026-08-05 21:27 UTC (permalink / raw)
To: amd-gfx; +Cc: Tishko.Araz, Eric Huang
new api will be used to workaround a HW scheduler issue
for 100% usage when gpu has no workload.
Signed-off-by: Eric Huang <jinhuieric.huang@amd.com>
---
drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c | 23 +++++++++++++++++++++++
drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h | 4 ++++
drivers/gpu/drm/amd/amdgpu/mes_v11_0.c | 5 ++++-
3 files changed, 31 insertions(+), 1 deletion(-)
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c
index b96f94e5169f..421d937c1188 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c
@@ -1140,6 +1140,29 @@ void amdgpu_mes_free_gang_ctx_index(struct amdgpu_mes *mes,
amdgpu_mes_unlock(mes);
}
+int amdgpu_mes_notify_unmap_queue(struct amdgpu_device *adev)
+{
+ struct mes_misc_op_input op_input = {0};
+ int r;
+
+ op_input.op = MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE;
+
+ if (!adev->mes.funcs->misc_op) {
+ dev_err(adev->dev, "mes notify unmap queue is not supported!\n");
+ r = -EINVAL;
+ goto error;
+ }
+
+ amdgpu_mes_lock(&adev->mes);
+ r = adev->mes.funcs->misc_op(&adev->mes, &op_input);
+ amdgpu_mes_unlock(&adev->mes);
+ if (r)
+ dev_err(adev->dev, "failed to notify unmap queue.\n");
+
+error:
+ return r;
+}
+
#if defined(CONFIG_DEBUG_FS)
static int amdgpu_debugfs_mes_event_log_show(struct seq_file *m, void *unused)
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h
index 977c057dcce8..5230b1ba1a46 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h
@@ -366,6 +366,7 @@ enum mes_misc_opcode {
MES_MISC_OP_WRM_REG_WR_WAIT,
MES_MISC_OP_SET_SHADER_DEBUGGER,
MES_MISC_OP_CHANGE_CONFIG,
+ MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE
};
struct mes_misc_op_input {
@@ -643,4 +644,7 @@ int amdgpu_mes_alloc_gang_ctx_index(struct amdgpu_mes *mes,
uint32_t *index);
void amdgpu_mes_free_gang_ctx_index(struct amdgpu_mes *mes,
uint32_t index);
+
+int amdgpu_mes_notify_unmap_queue(struct amdgpu_device *adev);
+
#endif /* __AMDGPU_MES_H__ */
diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
index a0b874c9ee7b..bc0ee6ccfca7 100644
--- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
@@ -957,7 +957,10 @@ static int mes_v11_0_misc_op(struct amdgpu_mes *mes,
misc_pkt.change_config.option.bits.limit_single_process =
input->change_config.option.limit_single_process;
break;
-
+ case MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE:
+ misc_pkt.opcode = MESAPI_MISC__NOTIFY_WORK_ON_UNMAPPED_QUEUE;
+ misc_pkt.queue_sch_level = AMD_PRIORITY_LEVEL_NORMAL;
+ break;
default:
drm_err(adev_to_drm(mes->adev), "unsupported misc op (%d)\n", input->op);
return -EINVAL;
--
2.34.1
^ permalink raw reply related [flat|nested] 5+ messages in thread
* [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue for gfx11
2026-08-05 21:27 [PATCH 1/2] drm/amdgpu: add new mes misc api for gfx11 Eric Huang
@ 2026-08-05 21:27 ` Eric Huang
2026-08-13 15:04 ` Eric Huang
2026-08-13 15:35 ` Alex Deucher
2026-08-13 15:34 ` [PATCH 1/2] drm/amdgpu: add new mes misc api " Alex Deucher
1 sibling, 2 replies; 5+ messages in thread
From: Eric Huang @ 2026-08-05 21:27 UTC (permalink / raw)
To: amd-gfx; +Cc: Tishko.Araz, Eric Huang
the issue only happens with oversubscription when gpu has no
workload, the root cause is mes oversubscription timer, so
disable mes timer and make a similar timer in kfd to resolve
the issue.
Signed-off-by: Eric Huang <jinhuieric.huang@amd.com>
---
drivers/gpu/drm/amd/amdgpu/mes_v11_0.c | 1 -
.../drm/amd/amdkfd/kfd_device_queue_manager.c | 39 ++++++++++++++++++-
.../drm/amd/amdkfd/kfd_device_queue_manager.h | 1 +
3 files changed, 39 insertions(+), 2 deletions(-)
diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
index bc0ee6ccfca7..40683db66a13 100644
--- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
@@ -1019,7 +1019,6 @@ static int mes_v11_0_set_hw_resources(struct amdgpu_mes *mes)
mes_set_hw_res_pkt.use_different_vmid_compute = 1;
mes_set_hw_res_pkt.enable_reg_active_poll = 1;
mes_set_hw_res_pkt.enable_level_process_quantum_check = 1;
- mes_set_hw_res_pkt.oversubscription_timer = 50;
if (adev->mes.use_rs64mem)
mes_set_hw_res_pkt.use_rs64mem_for_proc_gang_ctx = 1;
diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
index ea9d87450eae..350e0ce308d3 100644
--- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
+++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
@@ -47,6 +47,9 @@
/* See unmap_queues_cpsch() */
#define USE_DEFAULT_GRACE_PERIOD 0xffffffff
+/* Interval for notifying MES of work on unmapped queues during oversubscription */
+#define DQM_MES_UNMAP_NOTIFY_DELAY_MS 50
+
static int set_pasid_vmid_mapping(struct device_queue_manager *dqm,
u32 pasid, unsigned int vmid);
@@ -276,8 +279,16 @@ static int add_queue_mes(struct device_queue_manager *dqm, struct queue *q,
q->properties.doorbell_off);
dev_err(adev->dev, "MES might be in unrecoverable state, issue a GPU reset\n");
kfd_hws_hang(dqm);
+ return r;
}
+ /* GFX11: start notify timer only once when oversubscription begins */
+ if (KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
+ KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
+ dqm->active_cp_queue_count > get_cp_queues_num(dqm))
+ queue_delayed_work(system_wq, &dqm->notify_unmap_work,
+ msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
+
return r;
}
@@ -354,7 +365,16 @@ static void set_perfcount(struct device_queue_manager *dqm, int enable)
static int remove_queue_mes(struct device_queue_manager *dqm, struct queue *q,
struct qcm_process_device *qpd)
{
- return remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
+ int r = remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
+
+ /* GFX11: stop notify timer when oversubscription clears */
+ if (!r &&
+ KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
+ KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
+ dqm->active_cp_queue_count <= get_cp_queues_num(dqm))
+ cancel_delayed_work(&dqm->notify_unmap_work);
+
+ return r;
}
static int remove_all_kfd_queues_mes(struct device_queue_manager *dqm)
@@ -3088,6 +3108,20 @@ static void deallocate_hiq_sdma_mqd(struct kfd_node *dev,
amdgpu_amdkfd_free_kernel_mem(dev->adev, &mqd->mem);
}
+static void mes_notify_unmap_work_handler(struct work_struct *work)
+{
+ struct device_queue_manager *dqm =
+ container_of(work, struct device_queue_manager,
+ notify_unmap_work.work);
+
+ amdgpu_mes_notify_unmap_queue((struct amdgpu_device *)dqm->dev->adev);
+
+ /* Re-arm if still oversubscribed */
+ if (READ_ONCE(dqm->active_cp_queue_count) > get_cp_queues_num(dqm))
+ queue_delayed_work(system_wq, &dqm->notify_unmap_work,
+ msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
+}
+
struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
{
struct device_queue_manager *dqm;
@@ -3213,6 +3247,8 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
if (!dqm->ops.initialize(dqm)) {
init_waitqueue_head(&dqm->destroy_wait);
+ INIT_DELAYED_WORK(&dqm->notify_unmap_work,
+ mes_notify_unmap_work_handler);
return dqm;
}
@@ -3229,6 +3265,7 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
void device_queue_manager_uninit(struct device_queue_manager *dqm)
{
+ cancel_delayed_work_sync(&dqm->notify_unmap_work);
dqm->ops.stop(dqm);
dqm->ops.uninitialize(dqm);
if (!dqm->dev->kfd->shared_resources.enable_mes)
diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
index c9f9f7a87111..21cf3c16f3f9 100644
--- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
+++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
@@ -281,6 +281,7 @@ struct device_queue_manager {
uint32_t wait_times;
wait_queue_head_t destroy_wait;
+ struct delayed_work notify_unmap_work;
/* for per-queue reset support */
struct dqm_detect_hang_info *detect_hang_info;
--
2.34.1
^ permalink raw reply related [flat|nested] 5+ messages in thread
* Re: [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue for gfx11
2026-08-05 21:27 ` [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue " Eric Huang
@ 2026-08-13 15:04 ` Eric Huang
2026-08-13 15:35 ` Alex Deucher
1 sibling, 0 replies; 5+ messages in thread
From: Eric Huang @ 2026-08-13 15:04 UTC (permalink / raw)
To: amd-gfx
Ping...
On 2026-08-05 17:27, Eric Huang wrote:
> the issue only happens with oversubscription when gpu has no
> workload, the root cause is mes oversubscription timer, so
> disable mes timer and make a similar timer in kfd to resolve
> the issue.
>
> Signed-off-by: Eric Huang <jinhuieric.huang@amd.com>
> ---
> drivers/gpu/drm/amd/amdgpu/mes_v11_0.c | 1 -
> .../drm/amd/amdkfd/kfd_device_queue_manager.c | 39 ++++++++++++++++++-
> .../drm/amd/amdkfd/kfd_device_queue_manager.h | 1 +
> 3 files changed, 39 insertions(+), 2 deletions(-)
>
> diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> index bc0ee6ccfca7..40683db66a13 100644
> --- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> +++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> @@ -1019,7 +1019,6 @@ static int mes_v11_0_set_hw_resources(struct amdgpu_mes *mes)
> mes_set_hw_res_pkt.use_different_vmid_compute = 1;
> mes_set_hw_res_pkt.enable_reg_active_poll = 1;
> mes_set_hw_res_pkt.enable_level_process_quantum_check = 1;
> - mes_set_hw_res_pkt.oversubscription_timer = 50;
> if (adev->mes.use_rs64mem)
> mes_set_hw_res_pkt.use_rs64mem_for_proc_gang_ctx = 1;
>
> diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> index ea9d87450eae..350e0ce308d3 100644
> --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> @@ -47,6 +47,9 @@
> /* See unmap_queues_cpsch() */
> #define USE_DEFAULT_GRACE_PERIOD 0xffffffff
>
> +/* Interval for notifying MES of work on unmapped queues during oversubscription */
> +#define DQM_MES_UNMAP_NOTIFY_DELAY_MS 50
> +
> static int set_pasid_vmid_mapping(struct device_queue_manager *dqm,
> u32 pasid, unsigned int vmid);
>
> @@ -276,8 +279,16 @@ static int add_queue_mes(struct device_queue_manager *dqm, struct queue *q,
> q->properties.doorbell_off);
> dev_err(adev->dev, "MES might be in unrecoverable state, issue a GPU reset\n");
> kfd_hws_hang(dqm);
> + return r;
> }
>
> + /* GFX11: start notify timer only once when oversubscription begins */
> + if (KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
> + KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
> + dqm->active_cp_queue_count > get_cp_queues_num(dqm))
> + queue_delayed_work(system_wq, &dqm->notify_unmap_work,
> + msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
> +
> return r;
> }
>
> @@ -354,7 +365,16 @@ static void set_perfcount(struct device_queue_manager *dqm, int enable)
> static int remove_queue_mes(struct device_queue_manager *dqm, struct queue *q,
> struct qcm_process_device *qpd)
> {
> - return remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
> + int r = remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
> +
> + /* GFX11: stop notify timer when oversubscription clears */
> + if (!r &&
> + KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
> + KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
> + dqm->active_cp_queue_count <= get_cp_queues_num(dqm))
> + cancel_delayed_work(&dqm->notify_unmap_work);
> +
> + return r;
> }
>
> static int remove_all_kfd_queues_mes(struct device_queue_manager *dqm)
> @@ -3088,6 +3108,20 @@ static void deallocate_hiq_sdma_mqd(struct kfd_node *dev,
> amdgpu_amdkfd_free_kernel_mem(dev->adev, &mqd->mem);
> }
>
> +static void mes_notify_unmap_work_handler(struct work_struct *work)
> +{
> + struct device_queue_manager *dqm =
> + container_of(work, struct device_queue_manager,
> + notify_unmap_work.work);
> +
> + amdgpu_mes_notify_unmap_queue((struct amdgpu_device *)dqm->dev->adev);
> +
> + /* Re-arm if still oversubscribed */
> + if (READ_ONCE(dqm->active_cp_queue_count) > get_cp_queues_num(dqm))
> + queue_delayed_work(system_wq, &dqm->notify_unmap_work,
> + msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
> +}
> +
> struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
> {
> struct device_queue_manager *dqm;
> @@ -3213,6 +3247,8 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
>
> if (!dqm->ops.initialize(dqm)) {
> init_waitqueue_head(&dqm->destroy_wait);
> + INIT_DELAYED_WORK(&dqm->notify_unmap_work,
> + mes_notify_unmap_work_handler);
> return dqm;
> }
>
> @@ -3229,6 +3265,7 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
>
> void device_queue_manager_uninit(struct device_queue_manager *dqm)
> {
> + cancel_delayed_work_sync(&dqm->notify_unmap_work);
> dqm->ops.stop(dqm);
> dqm->ops.uninitialize(dqm);
> if (!dqm->dev->kfd->shared_resources.enable_mes)
> diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> index c9f9f7a87111..21cf3c16f3f9 100644
> --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> @@ -281,6 +281,7 @@ struct device_queue_manager {
> uint32_t wait_times;
>
> wait_queue_head_t destroy_wait;
> + struct delayed_work notify_unmap_work;
>
> /* for per-queue reset support */
> struct dqm_detect_hang_info *detect_hang_info;
^ permalink raw reply [flat|nested] 5+ messages in thread
* Re: [PATCH 1/2] drm/amdgpu: add new mes misc api for gfx11
2026-08-05 21:27 [PATCH 1/2] drm/amdgpu: add new mes misc api for gfx11 Eric Huang
2026-08-05 21:27 ` [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue " Eric Huang
@ 2026-08-13 15:34 ` Alex Deucher
1 sibling, 0 replies; 5+ messages in thread
From: Alex Deucher @ 2026-08-13 15:34 UTC (permalink / raw)
To: Eric Huang; +Cc: amd-gfx, Tishko.Araz
On Wed, Aug 5, 2026 at 6:23 PM Eric Huang <jinhuieric.huang@amd.com> wrote:
>
> new api will be used to workaround a HW scheduler issue
> for 100% usage when gpu has no workload.
>
> Signed-off-by: Eric Huang <jinhuieric.huang@amd.com>
> ---
> drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c | 23 +++++++++++++++++++++++
> drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h | 4 ++++
> drivers/gpu/drm/amd/amdgpu/mes_v11_0.c | 5 ++++-
> 3 files changed, 31 insertions(+), 1 deletion(-)
>
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c
> index b96f94e5169f..421d937c1188 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c
> @@ -1140,6 +1140,29 @@ void amdgpu_mes_free_gang_ctx_index(struct amdgpu_mes *mes,
> amdgpu_mes_unlock(mes);
> }
>
> +int amdgpu_mes_notify_unmap_queue(struct amdgpu_device *adev)
> +{
> + struct mes_misc_op_input op_input = {0};
> + int r;
> +
> + op_input.op = MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE;
> +
> + if (!adev->mes.funcs->misc_op) {
> + dev_err(adev->dev, "mes notify unmap queue is not supported!\n");
> + r = -EINVAL;
> + goto error;
> + }
> +
> + amdgpu_mes_lock(&adev->mes);
> + r = adev->mes.funcs->misc_op(&adev->mes, &op_input);
> + amdgpu_mes_unlock(&adev->mes);
> + if (r)
> + dev_err(adev->dev, "failed to notify unmap queue.\n");
> +
> +error:
> + return r;
> +}
> +
> #if defined(CONFIG_DEBUG_FS)
>
> static int amdgpu_debugfs_mes_event_log_show(struct seq_file *m, void *unused)
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h
> index 977c057dcce8..5230b1ba1a46 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h
> @@ -366,6 +366,7 @@ enum mes_misc_opcode {
> MES_MISC_OP_WRM_REG_WR_WAIT,
> MES_MISC_OP_SET_SHADER_DEBUGGER,
> MES_MISC_OP_CHANGE_CONFIG,
> + MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE
What's the minimum fw version which supports this? We need to check
so we only use it when supported.
Alex
> };
>
> struct mes_misc_op_input {
> @@ -643,4 +644,7 @@ int amdgpu_mes_alloc_gang_ctx_index(struct amdgpu_mes *mes,
> uint32_t *index);
> void amdgpu_mes_free_gang_ctx_index(struct amdgpu_mes *mes,
> uint32_t index);
> +
> +int amdgpu_mes_notify_unmap_queue(struct amdgpu_device *adev);
> +
> #endif /* __AMDGPU_MES_H__ */
> diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> index a0b874c9ee7b..bc0ee6ccfca7 100644
> --- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> +++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> @@ -957,7 +957,10 @@ static int mes_v11_0_misc_op(struct amdgpu_mes *mes,
> misc_pkt.change_config.option.bits.limit_single_process =
> input->change_config.option.limit_single_process;
> break;
> -
> + case MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE:
> + misc_pkt.opcode = MESAPI_MISC__NOTIFY_WORK_ON_UNMAPPED_QUEUE;
> + misc_pkt.queue_sch_level = AMD_PRIORITY_LEVEL_NORMAL;
> + break;
> default:
> drm_err(adev_to_drm(mes->adev), "unsupported misc op (%d)\n", input->op);
> return -EINVAL;
> --
> 2.34.1
>
^ permalink raw reply [flat|nested] 5+ messages in thread
* Re: [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue for gfx11
2026-08-05 21:27 ` [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue " Eric Huang
2026-08-13 15:04 ` Eric Huang
@ 2026-08-13 15:35 ` Alex Deucher
1 sibling, 0 replies; 5+ messages in thread
From: Alex Deucher @ 2026-08-13 15:35 UTC (permalink / raw)
To: Eric Huang; +Cc: amd-gfx, Tishko.Araz
On Wed, Aug 5, 2026 at 6:23 PM Eric Huang <jinhuieric.huang@amd.com> wrote:
>
> the issue only happens with oversubscription when gpu has no
> workload, the root cause is mes oversubscription timer, so
> disable mes timer and make a similar timer in kfd to resolve
> the issue.
This should only be enabled on FW versions which support it.
Additionally, we need similar treatment for KGD userqs if we make this
change.
Alex
>
> Signed-off-by: Eric Huang <jinhuieric.huang@amd.com>
> ---
> drivers/gpu/drm/amd/amdgpu/mes_v11_0.c | 1 -
> .../drm/amd/amdkfd/kfd_device_queue_manager.c | 39 ++++++++++++++++++-
> .../drm/amd/amdkfd/kfd_device_queue_manager.h | 1 +
> 3 files changed, 39 insertions(+), 2 deletions(-)
>
> diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> index bc0ee6ccfca7..40683db66a13 100644
> --- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> +++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> @@ -1019,7 +1019,6 @@ static int mes_v11_0_set_hw_resources(struct amdgpu_mes *mes)
> mes_set_hw_res_pkt.use_different_vmid_compute = 1;
> mes_set_hw_res_pkt.enable_reg_active_poll = 1;
> mes_set_hw_res_pkt.enable_level_process_quantum_check = 1;
> - mes_set_hw_res_pkt.oversubscription_timer = 50;
> if (adev->mes.use_rs64mem)
> mes_set_hw_res_pkt.use_rs64mem_for_proc_gang_ctx = 1;
>
> diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> index ea9d87450eae..350e0ce308d3 100644
> --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> @@ -47,6 +47,9 @@
> /* See unmap_queues_cpsch() */
> #define USE_DEFAULT_GRACE_PERIOD 0xffffffff
>
> +/* Interval for notifying MES of work on unmapped queues during oversubscription */
> +#define DQM_MES_UNMAP_NOTIFY_DELAY_MS 50
> +
> static int set_pasid_vmid_mapping(struct device_queue_manager *dqm,
> u32 pasid, unsigned int vmid);
>
> @@ -276,8 +279,16 @@ static int add_queue_mes(struct device_queue_manager *dqm, struct queue *q,
> q->properties.doorbell_off);
> dev_err(adev->dev, "MES might be in unrecoverable state, issue a GPU reset\n");
> kfd_hws_hang(dqm);
> + return r;
> }
>
> + /* GFX11: start notify timer only once when oversubscription begins */
> + if (KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
> + KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
> + dqm->active_cp_queue_count > get_cp_queues_num(dqm))
> + queue_delayed_work(system_wq, &dqm->notify_unmap_work,
> + msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
> +
> return r;
> }
>
> @@ -354,7 +365,16 @@ static void set_perfcount(struct device_queue_manager *dqm, int enable)
> static int remove_queue_mes(struct device_queue_manager *dqm, struct queue *q,
> struct qcm_process_device *qpd)
> {
> - return remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
> + int r = remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
> +
> + /* GFX11: stop notify timer when oversubscription clears */
> + if (!r &&
> + KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
> + KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
> + dqm->active_cp_queue_count <= get_cp_queues_num(dqm))
> + cancel_delayed_work(&dqm->notify_unmap_work);
> +
> + return r;
> }
>
> static int remove_all_kfd_queues_mes(struct device_queue_manager *dqm)
> @@ -3088,6 +3108,20 @@ static void deallocate_hiq_sdma_mqd(struct kfd_node *dev,
> amdgpu_amdkfd_free_kernel_mem(dev->adev, &mqd->mem);
> }
>
> +static void mes_notify_unmap_work_handler(struct work_struct *work)
> +{
> + struct device_queue_manager *dqm =
> + container_of(work, struct device_queue_manager,
> + notify_unmap_work.work);
> +
> + amdgpu_mes_notify_unmap_queue((struct amdgpu_device *)dqm->dev->adev);
> +
> + /* Re-arm if still oversubscribed */
> + if (READ_ONCE(dqm->active_cp_queue_count) > get_cp_queues_num(dqm))
> + queue_delayed_work(system_wq, &dqm->notify_unmap_work,
> + msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
> +}
> +
> struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
> {
> struct device_queue_manager *dqm;
> @@ -3213,6 +3247,8 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
>
> if (!dqm->ops.initialize(dqm)) {
> init_waitqueue_head(&dqm->destroy_wait);
> + INIT_DELAYED_WORK(&dqm->notify_unmap_work,
> + mes_notify_unmap_work_handler);
> return dqm;
> }
>
> @@ -3229,6 +3265,7 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
>
> void device_queue_manager_uninit(struct device_queue_manager *dqm)
> {
> + cancel_delayed_work_sync(&dqm->notify_unmap_work);
> dqm->ops.stop(dqm);
> dqm->ops.uninitialize(dqm);
> if (!dqm->dev->kfd->shared_resources.enable_mes)
> diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> index c9f9f7a87111..21cf3c16f3f9 100644
> --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> @@ -281,6 +281,7 @@ struct device_queue_manager {
> uint32_t wait_times;
>
> wait_queue_head_t destroy_wait;
> + struct delayed_work notify_unmap_work;
>
> /* for per-queue reset support */
> struct dqm_detect_hang_info *detect_hang_info;
> --
> 2.34.1
>
^ permalink raw reply [flat|nested] 5+ messages in thread
end of thread, other threads:[~2026-08-13 15:35 UTC | newest]
Thread overview: 5+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-08-05 21:27 [PATCH 1/2] drm/amdgpu: add new mes misc api for gfx11 Eric Huang
2026-08-05 21:27 ` [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue " Eric Huang
2026-08-13 15:04 ` Eric Huang
2026-08-13 15:35 ` Alex Deucher
2026-08-13 15:34 ` [PATCH 1/2] drm/amdgpu: add new mes misc api " Alex Deucher
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.