* [PATCH 1/2] drm/amdgpu: add new mes misc api for gfx11 @ 2026-08-05 21:27 Eric Huang 2026-08-05 21:27 ` [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue " Eric Huang 2026-08-13 15:34 ` [PATCH 1/2] drm/amdgpu: add new mes misc api " Alex Deucher 0 siblings, 2 replies; 5+ messages in thread From: Eric Huang @ 2026-08-05 21:27 UTC (permalink / raw) To: amd-gfx; +Cc: Tishko.Araz, Eric Huang new api will be used to workaround a HW scheduler issue for 100% usage when gpu has no workload. Signed-off-by: Eric Huang <jinhuieric.huang@amd.com> --- drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c | 23 +++++++++++++++++++++++ drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h | 4 ++++ drivers/gpu/drm/amd/amdgpu/mes_v11_0.c | 5 ++++- 3 files changed, 31 insertions(+), 1 deletion(-) diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c index b96f94e5169f..421d937c1188 100644 --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c @@ -1140,6 +1140,29 @@ void amdgpu_mes_free_gang_ctx_index(struct amdgpu_mes *mes, amdgpu_mes_unlock(mes); } +int amdgpu_mes_notify_unmap_queue(struct amdgpu_device *adev) +{ + struct mes_misc_op_input op_input = {0}; + int r; + + op_input.op = MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE; + + if (!adev->mes.funcs->misc_op) { + dev_err(adev->dev, "mes notify unmap queue is not supported!\n"); + r = -EINVAL; + goto error; + } + + amdgpu_mes_lock(&adev->mes); + r = adev->mes.funcs->misc_op(&adev->mes, &op_input); + amdgpu_mes_unlock(&adev->mes); + if (r) + dev_err(adev->dev, "failed to notify unmap queue.\n"); + +error: + return r; +} + #if defined(CONFIG_DEBUG_FS) static int amdgpu_debugfs_mes_event_log_show(struct seq_file *m, void *unused) diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h index 977c057dcce8..5230b1ba1a46 100644 --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h @@ -366,6 +366,7 @@ enum mes_misc_opcode { MES_MISC_OP_WRM_REG_WR_WAIT, MES_MISC_OP_SET_SHADER_DEBUGGER, MES_MISC_OP_CHANGE_CONFIG, + MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE }; struct mes_misc_op_input { @@ -643,4 +644,7 @@ int amdgpu_mes_alloc_gang_ctx_index(struct amdgpu_mes *mes, uint32_t *index); void amdgpu_mes_free_gang_ctx_index(struct amdgpu_mes *mes, uint32_t index); + +int amdgpu_mes_notify_unmap_queue(struct amdgpu_device *adev); + #endif /* __AMDGPU_MES_H__ */ diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c index a0b874c9ee7b..bc0ee6ccfca7 100644 --- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c +++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c @@ -957,7 +957,10 @@ static int mes_v11_0_misc_op(struct amdgpu_mes *mes, misc_pkt.change_config.option.bits.limit_single_process = input->change_config.option.limit_single_process; break; - + case MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE: + misc_pkt.opcode = MESAPI_MISC__NOTIFY_WORK_ON_UNMAPPED_QUEUE; + misc_pkt.queue_sch_level = AMD_PRIORITY_LEVEL_NORMAL; + break; default: drm_err(adev_to_drm(mes->adev), "unsupported misc op (%d)\n", input->op); return -EINVAL; -- 2.34.1 ^ permalink raw reply related [flat|nested] 5+ messages in thread
* [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue for gfx11 2026-08-05 21:27 [PATCH 1/2] drm/amdgpu: add new mes misc api for gfx11 Eric Huang @ 2026-08-05 21:27 ` Eric Huang 2026-08-13 15:04 ` Eric Huang 2026-08-13 15:35 ` Alex Deucher 2026-08-13 15:34 ` [PATCH 1/2] drm/amdgpu: add new mes misc api " Alex Deucher 1 sibling, 2 replies; 5+ messages in thread From: Eric Huang @ 2026-08-05 21:27 UTC (permalink / raw) To: amd-gfx; +Cc: Tishko.Araz, Eric Huang the issue only happens with oversubscription when gpu has no workload, the root cause is mes oversubscription timer, so disable mes timer and make a similar timer in kfd to resolve the issue. Signed-off-by: Eric Huang <jinhuieric.huang@amd.com> --- drivers/gpu/drm/amd/amdgpu/mes_v11_0.c | 1 - .../drm/amd/amdkfd/kfd_device_queue_manager.c | 39 ++++++++++++++++++- .../drm/amd/amdkfd/kfd_device_queue_manager.h | 1 + 3 files changed, 39 insertions(+), 2 deletions(-) diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c index bc0ee6ccfca7..40683db66a13 100644 --- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c +++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c @@ -1019,7 +1019,6 @@ static int mes_v11_0_set_hw_resources(struct amdgpu_mes *mes) mes_set_hw_res_pkt.use_different_vmid_compute = 1; mes_set_hw_res_pkt.enable_reg_active_poll = 1; mes_set_hw_res_pkt.enable_level_process_quantum_check = 1; - mes_set_hw_res_pkt.oversubscription_timer = 50; if (adev->mes.use_rs64mem) mes_set_hw_res_pkt.use_rs64mem_for_proc_gang_ctx = 1; diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c index ea9d87450eae..350e0ce308d3 100644 --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c @@ -47,6 +47,9 @@ /* See unmap_queues_cpsch() */ #define USE_DEFAULT_GRACE_PERIOD 0xffffffff +/* Interval for notifying MES of work on unmapped queues during oversubscription */ +#define DQM_MES_UNMAP_NOTIFY_DELAY_MS 50 + static int set_pasid_vmid_mapping(struct device_queue_manager *dqm, u32 pasid, unsigned int vmid); @@ -276,8 +279,16 @@ static int add_queue_mes(struct device_queue_manager *dqm, struct queue *q, q->properties.doorbell_off); dev_err(adev->dev, "MES might be in unrecoverable state, issue a GPU reset\n"); kfd_hws_hang(dqm); + return r; } + /* GFX11: start notify timer only once when oversubscription begins */ + if (KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) && + KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) && + dqm->active_cp_queue_count > get_cp_queues_num(dqm)) + queue_delayed_work(system_wq, &dqm->notify_unmap_work, + msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS)); + return r; } @@ -354,7 +365,16 @@ static void set_perfcount(struct device_queue_manager *dqm, int enable) static int remove_queue_mes(struct device_queue_manager *dqm, struct queue *q, struct qcm_process_device *qpd) { - return remove_queue_mes_on_reset_option(dqm, q, qpd, false, false); + int r = remove_queue_mes_on_reset_option(dqm, q, qpd, false, false); + + /* GFX11: stop notify timer when oversubscription clears */ + if (!r && + KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) && + KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) && + dqm->active_cp_queue_count <= get_cp_queues_num(dqm)) + cancel_delayed_work(&dqm->notify_unmap_work); + + return r; } static int remove_all_kfd_queues_mes(struct device_queue_manager *dqm) @@ -3088,6 +3108,20 @@ static void deallocate_hiq_sdma_mqd(struct kfd_node *dev, amdgpu_amdkfd_free_kernel_mem(dev->adev, &mqd->mem); } +static void mes_notify_unmap_work_handler(struct work_struct *work) +{ + struct device_queue_manager *dqm = + container_of(work, struct device_queue_manager, + notify_unmap_work.work); + + amdgpu_mes_notify_unmap_queue((struct amdgpu_device *)dqm->dev->adev); + + /* Re-arm if still oversubscribed */ + if (READ_ONCE(dqm->active_cp_queue_count) > get_cp_queues_num(dqm)) + queue_delayed_work(system_wq, &dqm->notify_unmap_work, + msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS)); +} + struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev) { struct device_queue_manager *dqm; @@ -3213,6 +3247,8 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev) if (!dqm->ops.initialize(dqm)) { init_waitqueue_head(&dqm->destroy_wait); + INIT_DELAYED_WORK(&dqm->notify_unmap_work, + mes_notify_unmap_work_handler); return dqm; } @@ -3229,6 +3265,7 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev) void device_queue_manager_uninit(struct device_queue_manager *dqm) { + cancel_delayed_work_sync(&dqm->notify_unmap_work); dqm->ops.stop(dqm); dqm->ops.uninitialize(dqm); if (!dqm->dev->kfd->shared_resources.enable_mes) diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h index c9f9f7a87111..21cf3c16f3f9 100644 --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h @@ -281,6 +281,7 @@ struct device_queue_manager { uint32_t wait_times; wait_queue_head_t destroy_wait; + struct delayed_work notify_unmap_work; /* for per-queue reset support */ struct dqm_detect_hang_info *detect_hang_info; -- 2.34.1 ^ permalink raw reply related [flat|nested] 5+ messages in thread
* Re: [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue for gfx11 2026-08-05 21:27 ` [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue " Eric Huang @ 2026-08-13 15:04 ` Eric Huang 2026-08-13 15:35 ` Alex Deucher 1 sibling, 0 replies; 5+ messages in thread From: Eric Huang @ 2026-08-13 15:04 UTC (permalink / raw) To: amd-gfx Ping... On 2026-08-05 17:27, Eric Huang wrote: > the issue only happens with oversubscription when gpu has no > workload, the root cause is mes oversubscription timer, so > disable mes timer and make a similar timer in kfd to resolve > the issue. > > Signed-off-by: Eric Huang <jinhuieric.huang@amd.com> > --- > drivers/gpu/drm/amd/amdgpu/mes_v11_0.c | 1 - > .../drm/amd/amdkfd/kfd_device_queue_manager.c | 39 ++++++++++++++++++- > .../drm/amd/amdkfd/kfd_device_queue_manager.h | 1 + > 3 files changed, 39 insertions(+), 2 deletions(-) > > diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c > index bc0ee6ccfca7..40683db66a13 100644 > --- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c > +++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c > @@ -1019,7 +1019,6 @@ static int mes_v11_0_set_hw_resources(struct amdgpu_mes *mes) > mes_set_hw_res_pkt.use_different_vmid_compute = 1; > mes_set_hw_res_pkt.enable_reg_active_poll = 1; > mes_set_hw_res_pkt.enable_level_process_quantum_check = 1; > - mes_set_hw_res_pkt.oversubscription_timer = 50; > if (adev->mes.use_rs64mem) > mes_set_hw_res_pkt.use_rs64mem_for_proc_gang_ctx = 1; > > diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c > index ea9d87450eae..350e0ce308d3 100644 > --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c > +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c > @@ -47,6 +47,9 @@ > /* See unmap_queues_cpsch() */ > #define USE_DEFAULT_GRACE_PERIOD 0xffffffff > > +/* Interval for notifying MES of work on unmapped queues during oversubscription */ > +#define DQM_MES_UNMAP_NOTIFY_DELAY_MS 50 > + > static int set_pasid_vmid_mapping(struct device_queue_manager *dqm, > u32 pasid, unsigned int vmid); > > @@ -276,8 +279,16 @@ static int add_queue_mes(struct device_queue_manager *dqm, struct queue *q, > q->properties.doorbell_off); > dev_err(adev->dev, "MES might be in unrecoverable state, issue a GPU reset\n"); > kfd_hws_hang(dqm); > + return r; > } > > + /* GFX11: start notify timer only once when oversubscription begins */ > + if (KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) && > + KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) && > + dqm->active_cp_queue_count > get_cp_queues_num(dqm)) > + queue_delayed_work(system_wq, &dqm->notify_unmap_work, > + msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS)); > + > return r; > } > > @@ -354,7 +365,16 @@ static void set_perfcount(struct device_queue_manager *dqm, int enable) > static int remove_queue_mes(struct device_queue_manager *dqm, struct queue *q, > struct qcm_process_device *qpd) > { > - return remove_queue_mes_on_reset_option(dqm, q, qpd, false, false); > + int r = remove_queue_mes_on_reset_option(dqm, q, qpd, false, false); > + > + /* GFX11: stop notify timer when oversubscription clears */ > + if (!r && > + KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) && > + KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) && > + dqm->active_cp_queue_count <= get_cp_queues_num(dqm)) > + cancel_delayed_work(&dqm->notify_unmap_work); > + > + return r; > } > > static int remove_all_kfd_queues_mes(struct device_queue_manager *dqm) > @@ -3088,6 +3108,20 @@ static void deallocate_hiq_sdma_mqd(struct kfd_node *dev, > amdgpu_amdkfd_free_kernel_mem(dev->adev, &mqd->mem); > } > > +static void mes_notify_unmap_work_handler(struct work_struct *work) > +{ > + struct device_queue_manager *dqm = > + container_of(work, struct device_queue_manager, > + notify_unmap_work.work); > + > + amdgpu_mes_notify_unmap_queue((struct amdgpu_device *)dqm->dev->adev); > + > + /* Re-arm if still oversubscribed */ > + if (READ_ONCE(dqm->active_cp_queue_count) > get_cp_queues_num(dqm)) > + queue_delayed_work(system_wq, &dqm->notify_unmap_work, > + msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS)); > +} > + > struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev) > { > struct device_queue_manager *dqm; > @@ -3213,6 +3247,8 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev) > > if (!dqm->ops.initialize(dqm)) { > init_waitqueue_head(&dqm->destroy_wait); > + INIT_DELAYED_WORK(&dqm->notify_unmap_work, > + mes_notify_unmap_work_handler); > return dqm; > } > > @@ -3229,6 +3265,7 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev) > > void device_queue_manager_uninit(struct device_queue_manager *dqm) > { > + cancel_delayed_work_sync(&dqm->notify_unmap_work); > dqm->ops.stop(dqm); > dqm->ops.uninitialize(dqm); > if (!dqm->dev->kfd->shared_resources.enable_mes) > diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h > index c9f9f7a87111..21cf3c16f3f9 100644 > --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h > +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h > @@ -281,6 +281,7 @@ struct device_queue_manager { > uint32_t wait_times; > > wait_queue_head_t destroy_wait; > + struct delayed_work notify_unmap_work; > > /* for per-queue reset support */ > struct dqm_detect_hang_info *detect_hang_info; ^ permalink raw reply [flat|nested] 5+ messages in thread
* Re: [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue for gfx11 2026-08-05 21:27 ` [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue " Eric Huang 2026-08-13 15:04 ` Eric Huang @ 2026-08-13 15:35 ` Alex Deucher 1 sibling, 0 replies; 5+ messages in thread From: Alex Deucher @ 2026-08-13 15:35 UTC (permalink / raw) To: Eric Huang; +Cc: amd-gfx, Tishko.Araz On Wed, Aug 5, 2026 at 6:23 PM Eric Huang <jinhuieric.huang@amd.com> wrote: > > the issue only happens with oversubscription when gpu has no > workload, the root cause is mes oversubscription timer, so > disable mes timer and make a similar timer in kfd to resolve > the issue. This should only be enabled on FW versions which support it. Additionally, we need similar treatment for KGD userqs if we make this change. Alex > > Signed-off-by: Eric Huang <jinhuieric.huang@amd.com> > --- > drivers/gpu/drm/amd/amdgpu/mes_v11_0.c | 1 - > .../drm/amd/amdkfd/kfd_device_queue_manager.c | 39 ++++++++++++++++++- > .../drm/amd/amdkfd/kfd_device_queue_manager.h | 1 + > 3 files changed, 39 insertions(+), 2 deletions(-) > > diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c > index bc0ee6ccfca7..40683db66a13 100644 > --- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c > +++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c > @@ -1019,7 +1019,6 @@ static int mes_v11_0_set_hw_resources(struct amdgpu_mes *mes) > mes_set_hw_res_pkt.use_different_vmid_compute = 1; > mes_set_hw_res_pkt.enable_reg_active_poll = 1; > mes_set_hw_res_pkt.enable_level_process_quantum_check = 1; > - mes_set_hw_res_pkt.oversubscription_timer = 50; > if (adev->mes.use_rs64mem) > mes_set_hw_res_pkt.use_rs64mem_for_proc_gang_ctx = 1; > > diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c > index ea9d87450eae..350e0ce308d3 100644 > --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c > +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c > @@ -47,6 +47,9 @@ > /* See unmap_queues_cpsch() */ > #define USE_DEFAULT_GRACE_PERIOD 0xffffffff > > +/* Interval for notifying MES of work on unmapped queues during oversubscription */ > +#define DQM_MES_UNMAP_NOTIFY_DELAY_MS 50 > + > static int set_pasid_vmid_mapping(struct device_queue_manager *dqm, > u32 pasid, unsigned int vmid); > > @@ -276,8 +279,16 @@ static int add_queue_mes(struct device_queue_manager *dqm, struct queue *q, > q->properties.doorbell_off); > dev_err(adev->dev, "MES might be in unrecoverable state, issue a GPU reset\n"); > kfd_hws_hang(dqm); > + return r; > } > > + /* GFX11: start notify timer only once when oversubscription begins */ > + if (KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) && > + KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) && > + dqm->active_cp_queue_count > get_cp_queues_num(dqm)) > + queue_delayed_work(system_wq, &dqm->notify_unmap_work, > + msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS)); > + > return r; > } > > @@ -354,7 +365,16 @@ static void set_perfcount(struct device_queue_manager *dqm, int enable) > static int remove_queue_mes(struct device_queue_manager *dqm, struct queue *q, > struct qcm_process_device *qpd) > { > - return remove_queue_mes_on_reset_option(dqm, q, qpd, false, false); > + int r = remove_queue_mes_on_reset_option(dqm, q, qpd, false, false); > + > + /* GFX11: stop notify timer when oversubscription clears */ > + if (!r && > + KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) && > + KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) && > + dqm->active_cp_queue_count <= get_cp_queues_num(dqm)) > + cancel_delayed_work(&dqm->notify_unmap_work); > + > + return r; > } > > static int remove_all_kfd_queues_mes(struct device_queue_manager *dqm) > @@ -3088,6 +3108,20 @@ static void deallocate_hiq_sdma_mqd(struct kfd_node *dev, > amdgpu_amdkfd_free_kernel_mem(dev->adev, &mqd->mem); > } > > +static void mes_notify_unmap_work_handler(struct work_struct *work) > +{ > + struct device_queue_manager *dqm = > + container_of(work, struct device_queue_manager, > + notify_unmap_work.work); > + > + amdgpu_mes_notify_unmap_queue((struct amdgpu_device *)dqm->dev->adev); > + > + /* Re-arm if still oversubscribed */ > + if (READ_ONCE(dqm->active_cp_queue_count) > get_cp_queues_num(dqm)) > + queue_delayed_work(system_wq, &dqm->notify_unmap_work, > + msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS)); > +} > + > struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev) > { > struct device_queue_manager *dqm; > @@ -3213,6 +3247,8 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev) > > if (!dqm->ops.initialize(dqm)) { > init_waitqueue_head(&dqm->destroy_wait); > + INIT_DELAYED_WORK(&dqm->notify_unmap_work, > + mes_notify_unmap_work_handler); > return dqm; > } > > @@ -3229,6 +3265,7 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev) > > void device_queue_manager_uninit(struct device_queue_manager *dqm) > { > + cancel_delayed_work_sync(&dqm->notify_unmap_work); > dqm->ops.stop(dqm); > dqm->ops.uninitialize(dqm); > if (!dqm->dev->kfd->shared_resources.enable_mes) > diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h > index c9f9f7a87111..21cf3c16f3f9 100644 > --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h > +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h > @@ -281,6 +281,7 @@ struct device_queue_manager { > uint32_t wait_times; > > wait_queue_head_t destroy_wait; > + struct delayed_work notify_unmap_work; > > /* for per-queue reset support */ > struct dqm_detect_hang_info *detect_hang_info; > -- > 2.34.1 > ^ permalink raw reply [flat|nested] 5+ messages in thread
* Re: [PATCH 1/2] drm/amdgpu: add new mes misc api for gfx11 2026-08-05 21:27 [PATCH 1/2] drm/amdgpu: add new mes misc api for gfx11 Eric Huang 2026-08-05 21:27 ` [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue " Eric Huang @ 2026-08-13 15:34 ` Alex Deucher 1 sibling, 0 replies; 5+ messages in thread From: Alex Deucher @ 2026-08-13 15:34 UTC (permalink / raw) To: Eric Huang; +Cc: amd-gfx, Tishko.Araz On Wed, Aug 5, 2026 at 6:23 PM Eric Huang <jinhuieric.huang@amd.com> wrote: > > new api will be used to workaround a HW scheduler issue > for 100% usage when gpu has no workload. > > Signed-off-by: Eric Huang <jinhuieric.huang@amd.com> > --- > drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c | 23 +++++++++++++++++++++++ > drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h | 4 ++++ > drivers/gpu/drm/amd/amdgpu/mes_v11_0.c | 5 ++++- > 3 files changed, 31 insertions(+), 1 deletion(-) > > diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c > index b96f94e5169f..421d937c1188 100644 > --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c > +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.c > @@ -1140,6 +1140,29 @@ void amdgpu_mes_free_gang_ctx_index(struct amdgpu_mes *mes, > amdgpu_mes_unlock(mes); > } > > +int amdgpu_mes_notify_unmap_queue(struct amdgpu_device *adev) > +{ > + struct mes_misc_op_input op_input = {0}; > + int r; > + > + op_input.op = MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE; > + > + if (!adev->mes.funcs->misc_op) { > + dev_err(adev->dev, "mes notify unmap queue is not supported!\n"); > + r = -EINVAL; > + goto error; > + } > + > + amdgpu_mes_lock(&adev->mes); > + r = adev->mes.funcs->misc_op(&adev->mes, &op_input); > + amdgpu_mes_unlock(&adev->mes); > + if (r) > + dev_err(adev->dev, "failed to notify unmap queue.\n"); > + > +error: > + return r; > +} > + > #if defined(CONFIG_DEBUG_FS) > > static int amdgpu_debugfs_mes_event_log_show(struct seq_file *m, void *unused) > diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h > index 977c057dcce8..5230b1ba1a46 100644 > --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h > +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_mes.h > @@ -366,6 +366,7 @@ enum mes_misc_opcode { > MES_MISC_OP_WRM_REG_WR_WAIT, > MES_MISC_OP_SET_SHADER_DEBUGGER, > MES_MISC_OP_CHANGE_CONFIG, > + MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE What's the minimum fw version which supports this? We need to check so we only use it when supported. Alex > }; > > struct mes_misc_op_input { > @@ -643,4 +644,7 @@ int amdgpu_mes_alloc_gang_ctx_index(struct amdgpu_mes *mes, > uint32_t *index); > void amdgpu_mes_free_gang_ctx_index(struct amdgpu_mes *mes, > uint32_t index); > + > +int amdgpu_mes_notify_unmap_queue(struct amdgpu_device *adev); > + > #endif /* __AMDGPU_MES_H__ */ > diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c > index a0b874c9ee7b..bc0ee6ccfca7 100644 > --- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c > +++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c > @@ -957,7 +957,10 @@ static int mes_v11_0_misc_op(struct amdgpu_mes *mes, > misc_pkt.change_config.option.bits.limit_single_process = > input->change_config.option.limit_single_process; > break; > - > + case MES_MISC_NOTIFY_WORK_ON_UNMAPPED_QUEUE: > + misc_pkt.opcode = MESAPI_MISC__NOTIFY_WORK_ON_UNMAPPED_QUEUE; > + misc_pkt.queue_sch_level = AMD_PRIORITY_LEVEL_NORMAL; > + break; > default: > drm_err(adev_to_drm(mes->adev), "unsupported misc op (%d)\n", input->op); > return -EINVAL; > -- > 2.34.1 > ^ permalink raw reply [flat|nested] 5+ messages in thread
end of thread, other threads:[~2026-08-13 15:35 UTC | newest] Thread overview: 5+ messages (download: mbox.gz follow: Atom feed -- links below jump to the message on this page -- 2026-08-05 21:27 [PATCH 1/2] drm/amdgpu: add new mes misc api for gfx11 Eric Huang 2026-08-05 21:27 ` [PATCH 2/2] drm/amdkfd: workaround 100% gpu usage issue " Eric Huang 2026-08-13 15:04 ` Eric Huang 2026-08-13 15:35 ` Alex Deucher 2026-08-13 15:34 ` [PATCH 1/2] drm/amdgpu: add new mes misc api " Alex Deucher
This is an external index of several public inboxes, see mirroring instructions on how to clone and mirror all data and code used by this external index.