From: Mario Limonciello <superm1@kernel.org>
To: Eric Huang <jinhuieric.huang@amd.com>, amd-gfx@lists.freedesktop.org
Cc: Tishko.Araz@amd.com, Alexander.Deucher@amd.com
Subject: Re: [PATCH 2/3] drm/amdkfd: workaround 100% gpu usage issue for gfx11
Date: Fri, 21 Aug 2026 16:31:01 -0500 [thread overview]
Message-ID: <478cee51-9b47-406e-9993-a116584afca5@kernel.org> (raw)
In-Reply-To: <20260819201443.282689-2-jinhuieric.huang@amd.com>
On 8/19/26 15:14, Eric Huang wrote:
> the issue only happens with oversubscription when gpu has no
> workload, the root cause is mes oversubscription timer, so
> disable mes timer and make a similar timer in kfd to resolve
> the issue.
>
> Signed-off-by: Eric Huang <jinhuieric.huang@amd.com>
Reviewed-by: Mario Limonciello (AMD) <superm1@kernel.org>
> ---
> drivers/gpu/drm/amd/amdgpu/mes_v11_0.c | 1 -
> .../drm/amd/amdkfd/kfd_device_queue_manager.c | 39 ++++++++++++++++++-
> .../drm/amd/amdkfd/kfd_device_queue_manager.h | 1 +
> 3 files changed, 39 insertions(+), 2 deletions(-)
>
> diff --git a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> index df71e9447b35..33ff1afd7c4c 100644
> --- a/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> +++ b/drivers/gpu/drm/amd/amdgpu/mes_v11_0.c
> @@ -1024,7 +1024,6 @@ static int mes_v11_0_set_hw_resources(struct amdgpu_mes *mes)
> mes_set_hw_res_pkt.use_different_vmid_compute = 1;
> mes_set_hw_res_pkt.enable_reg_active_poll = 1;
> mes_set_hw_res_pkt.enable_level_process_quantum_check = 1;
> - mes_set_hw_res_pkt.oversubscription_timer = 50;
> if (adev->mes.use_rs64mem)
> mes_set_hw_res_pkt.use_rs64mem_for_proc_gang_ctx = 1;
>
> diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> index a23384571193..45e039c9f31b 100644
> --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.c
> @@ -47,6 +47,9 @@
> /* See unmap_queues_cpsch() */
> #define USE_DEFAULT_GRACE_PERIOD 0xffffffff
>
> +/* Interval for notifying MES of work on unmapped queues during oversubscription */
> +#define DQM_MES_UNMAP_NOTIFY_DELAY_MS 50
> +
> static int set_pasid_vmid_mapping(struct device_queue_manager *dqm,
> u32 pasid, unsigned int vmid);
>
> @@ -276,8 +279,16 @@ static int add_queue_mes(struct device_queue_manager *dqm, struct queue *q,
> q->properties.doorbell_off);
> dev_err(adev->dev, "MES might be in unrecoverable state, issue a GPU reset\n");
> kfd_hws_hang(dqm);
> + return r;
> }
>
> + /* GFX11: start notify timer only once when oversubscription begins */
> + if (KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
> + KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
> + dqm->active_cp_queue_count > get_cp_queues_num(dqm))
> + queue_delayed_work(system_wq, &dqm->notify_unmap_work,
> + msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
> +
> return r;
> }
>
> @@ -354,7 +365,16 @@ static void set_perfcount(struct device_queue_manager *dqm, int enable)
> static int remove_queue_mes(struct device_queue_manager *dqm, struct queue *q,
> struct qcm_process_device *qpd)
> {
> - return remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
> + int r = remove_queue_mes_on_reset_option(dqm, q, qpd, false, false);
> +
> + /* GFX11: stop notify timer when oversubscription clears */
> + if (!r &&
> + KFD_GC_VERSION(dqm->dev) >= IP_VERSION(11, 0, 0) &&
> + KFD_GC_VERSION(dqm->dev) < IP_VERSION(12, 0, 0) &&
> + dqm->active_cp_queue_count <= get_cp_queues_num(dqm))
> + cancel_delayed_work(&dqm->notify_unmap_work);
> +
> + return r;
> }
>
> static int remove_all_kfd_queues_mes(struct device_queue_manager *dqm)
> @@ -3194,6 +3214,20 @@ static void deallocate_hiq_sdma_mqd(struct kfd_node *dev,
> amdgpu_amdkfd_free_kernel_mem(dev->adev, &mqd->mem);
> }
>
> +static void mes_notify_unmap_work_handler(struct work_struct *work)
> +{
> + struct device_queue_manager *dqm =
> + container_of(work, struct device_queue_manager,
> + notify_unmap_work.work);
> +
> + amdgpu_mes_notify_unmap_queue((struct amdgpu_device *)dqm->dev->adev);
> +
> + /* Re-arm if still oversubscribed */
> + if (READ_ONCE(dqm->active_cp_queue_count) > get_cp_queues_num(dqm))
> + queue_delayed_work(system_wq, &dqm->notify_unmap_work,
> + msecs_to_jiffies(DQM_MES_UNMAP_NOTIFY_DELAY_MS));
> +}
> +
> struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
> {
> struct device_queue_manager *dqm;
> @@ -3319,6 +3353,8 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
>
> if (!dqm->ops.initialize(dqm)) {
> init_waitqueue_head(&dqm->destroy_wait);
> + INIT_DELAYED_WORK(&dqm->notify_unmap_work,
> + mes_notify_unmap_work_handler);
> return dqm;
> }
>
> @@ -3335,6 +3371,7 @@ struct device_queue_manager *device_queue_manager_init(struct kfd_node *dev)
>
> void device_queue_manager_uninit(struct device_queue_manager *dqm)
> {
> + cancel_delayed_work_sync(&dqm->notify_unmap_work);
> dqm->ops.stop(dqm);
> dqm->ops.uninitialize(dqm);
> if (!dqm->dev->kfd->shared_resources.enable_mes)
> diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> index c9f9f7a87111..21cf3c16f3f9 100644
> --- a/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> +++ b/drivers/gpu/drm/amd/amdkfd/kfd_device_queue_manager.h
> @@ -281,6 +281,7 @@ struct device_queue_manager {
> uint32_t wait_times;
>
> wait_queue_head_t destroy_wait;
> + struct delayed_work notify_unmap_work;
>
> /* for per-queue reset support */
> struct dqm_detect_hang_info *detect_hang_info;
next prev parent reply other threads:[~2026-08-21 21:31 UTC|newest]
Thread overview: 6+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-08-19 20:14 [PATCH 1/3] drm/amdgpu: add new mes misc api for gfx11 Eric Huang
2026-08-19 20:14 ` [PATCH 2/3] drm/amdkfd: workaround 100% gpu usage issue " Eric Huang
2026-08-21 21:31 ` Mario Limonciello [this message]
2026-08-19 20:14 ` [PATCH 3/3] drm/amdgpu/userq: create the same workaround of oversubscription timer Eric Huang
2026-08-21 21:32 ` Mario Limonciello
2026-08-21 21:30 ` [PATCH 1/3] drm/amdgpu: add new mes misc api for gfx11 Mario Limonciello
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=478cee51-9b47-406e-9993-a116584afca5@kernel.org \
--to=superm1@kernel.org \
--cc=Alexander.Deucher@amd.com \
--cc=Tishko.Araz@amd.com \
--cc=amd-gfx@lists.freedesktop.org \
--cc=jinhuieric.huang@amd.com \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.