* [PATCH] drm/kfd: Add CU occupancy support to GFX12.1
@ 2026-08-17 21:49 David Belanger
2026-08-18 2:50 ` Somasekharan, Sreekant
0 siblings, 1 reply; 2+ messages in thread
From: David Belanger @ 2026-08-17 21:49 UTC (permalink / raw)
To: amd-gfx; +Cc: David Belanger
Port changes from GFX9 to GFX12.1 mostly as-is.
Minor changes to register access code.
Assisted-by: Claude:Sonnet 4.6
Signed-off-by: David Belanger <david.belanger@amd.com>
---
.../drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c | 150 +++++++++++++++++-
1 file changed, 149 insertions(+), 1 deletion(-)
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c
index 38ca1aea33b2f..b9a4a365e58ab 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c
@@ -371,6 +371,153 @@ static uint32_t kgd_gfx_v12_1_hqd_sdma_get_doorbell(struct amdgpu_device *adev,
return 0;
}
+static void lock_spi_csq_mutexes(struct amdgpu_device *adev)
+{
+ mutex_lock(&adev->srbm_mutex);
+ mutex_lock(&adev->grbm_idx_mutex);
+
+}
+
+static void unlock_spi_csq_mutexes(struct amdgpu_device *adev)
+{
+ mutex_unlock(&adev->grbm_idx_mutex);
+ mutex_unlock(&adev->srbm_mutex);
+}
+
+/**
+ * get_wave_count: Read device registers to get number of waves in flight for
+ * a particular queue. The method also returns the doorbell offset associated
+ * with the queue.
+ *
+ * @adev: Handle of device whose registers are to be read
+ * @queue_idx: Index of queue in the queue-map bit-field
+ * @queue_cnt: Stores the wave count and doorbell offset for an active queue
+ * @inst: xcc's instance number on a multi-XCC setup
+ */
+static void get_wave_count(struct amdgpu_device *adev, int queue_idx,
+ struct kfd_cu_occupancy *queue_cnt, uint32_t inst)
+{
+ int pipe_idx;
+ int queue_slot;
+ unsigned int reg_val;
+ unsigned int wave_cnt;
+ /*
+ * Program GRBM with appropriate MEID, PIPEID, QUEUEID and VMID
+ * parameters to read out waves in flight. Get VMID if there are
+ * non-zero waves in flight.
+ */
+ pipe_idx = queue_idx / adev->gfx.mec.num_queue_per_pipe;
+ queue_slot = queue_idx % adev->gfx.mec.num_queue_per_pipe;
+ amdgpu_gfx_select_me_pipe_q(adev, 1, pipe_idx, queue_slot, 0, inst);
+ reg_val = RREG32_SOC15_IP(GC, SOC15_REG_OFFSET(GC, GET_INST(GC, inst),
+ regSPI_CSQ_WF_ACTIVE_COUNT_0) + queue_slot);
+ wave_cnt = reg_val & SPI_CSQ_WF_ACTIVE_COUNT_0__COUNT_MASK;
+ if (wave_cnt != 0) {
+ queue_cnt->wave_cnt += wave_cnt;
+ queue_cnt->doorbell_off =
+ (RREG32_SOC15(GC, GET_INST(GC, inst), regCP_HQD_PQ_DOORBELL_CONTROL) &
+ CP_HQD_PQ_DOORBELL_CONTROL__DOORBELL_OFFSET_MASK) >>
+ CP_HQD_PQ_DOORBELL_CONTROL__DOORBELL_OFFSET__SHIFT;
+ }
+}
+
+/**
+ * kgd_gfx_v12_1_get_cu_occupancy: Reads relevant registers associated with
+ * each shader engine and aggregates the number of waves that are in flight
+ * for the process whose pasid is provided as a parameter. The process could
+ * have ZERO or more queues running and submitting waves to compute units.
+ *
+ * @adev: Handle of device from which to get number of waves in flight
+ * @cu_occupancy: Array that gets filled with wave_cnt and doorbell offset
+ * for comparison later.
+ * @max_waves_per_cu: Output parameter updated with maximum number of waves
+ * possible per Compute Unit
+ * @inst: xcc's instance number on a multi-XCC setup
+ *
+ * Note: It's possible that the device has too many queues (oversubscription)
+ * in which case a VMID could be remapped to a different PASID. This could lead
+ * to an inaccurate wave count. Following is a high-level sequence:
+ * Time T1: vmid = getVmid(); vmid is associated with Pasid P1
+ * Time T2: passId = getPasId(vmid); vmid is associated with Pasid P2
+ * In the sequence above wave count obtained from time T1 will be incorrectly
+ * lost or added to total wave count.
+ *
+ * The registers that provide the waves in flight are:
+ *
+ * SPI_CSQ_WF_ACTIVE_STATUS - bit-map of queues per pipe. The bit is ON if a
+ * queue is slotted, OFF if there is no queue. A process could have ZERO or
+ * more queues slotted and submitting waves to be run on compute units. Even
+ * when there is a queue it is possible there could be zero wave fronts, this
+ * can happen when queue is waiting on top-of-pipe events - e.g. waitRegMem
+ * command
+ *
+ * For each bit that is ON from above:
+ *
+ * Read (SPI_CSQ_WF_ACTIVE_COUNT_0 + queue_idx) register. It provides the
+ * number of waves that are in flight for the queue at specified index. The
+ * index ranges from 0 to 7.
+ *
+ * If non-zero waves are in flight, store the corresponding doorbell offset
+ * of the queue, along with the wave count.
+ *
+ * Determine if the queue belongs to the process by comparing the doorbell
+ * offset against the process's queues. If it matches, aggregate the wave
+ * count for the process.
+ *
+ * Reading registers referenced above involves programming GRBM appropriately
+ */
+static void kgd_gfx_v12_1_get_cu_occupancy(struct amdgpu_device *adev,
+ struct kfd_cu_occupancy *cu_occupancy,
+ int *max_waves_per_cu, uint32_t inst)
+{
+ int qidx;
+ int se_idx;
+ int se_cnt;
+ int queue_map;
+ int max_queue_cnt;
+ DECLARE_BITMAP(cp_queue_bitmap, AMDGPU_MAX_QUEUES);
+
+ lock_spi_csq_mutexes(adev);
+ amdgpu_gfx_select_me_pipe_q(adev, 1, 0, 0, 0, inst);
+
+ /*
+ * Iterate through the shader engines and arrays of the device
+ * to get number of waves in flight
+ */
+ bitmap_complement(cp_queue_bitmap, adev->gfx.mec_bitmap[0].queue_bitmap,
+ AMDGPU_MAX_QUEUES);
+ max_queue_cnt = adev->gfx.mec.num_pipe_per_mec *
+ adev->gfx.mec.num_queue_per_pipe;
+ se_cnt = adev->gfx.config.max_shader_engines;
+ for (se_idx = 0; se_idx < se_cnt; se_idx++) {
+ amdgpu_gfx_select_se_sh(adev, se_idx, 0, 0xffffffff, inst);
+ queue_map = RREG32_SOC15(GC, GET_INST(GC, inst),
+ regSPI_CSQ_WF_ACTIVE_STATUS);
+
+ for (qidx = 0; qidx < max_queue_cnt; qidx++) {
+ /* Skip queues that are not associated with
+ * compute functions
+ */
+ if (!test_bit(qidx, cp_queue_bitmap))
+ continue;
+
+ if (!(queue_map & (1 << qidx)))
+ continue;
+
+ /* Get number of waves in flight and aggregate them */
+ get_wave_count(adev, qidx, &cu_occupancy[qidx], inst);
+ }
+ }
+
+ amdgpu_gfx_select_se_sh(adev, 0xffffffff, 0xffffffff, 0xffffffff, inst);
+ amdgpu_gfx_select_me_pipe_q(adev, 0, 0, 0, 0, inst);
+ unlock_spi_csq_mutexes(adev);
+
+ /* Update the output parameters and return */
+ *max_waves_per_cu = adev->gfx.cu_info.simd_per_cu *
+ adev->gfx.cu_info.max_waves_per_simd;
+}
+
const struct kfd2kgd_calls gfx_v12_1_kfd2kgd = {
.init_interrupts = init_interrupts_v12_1,
.hqd_dump = hqd_dump_v12_1,
@@ -384,5 +531,6 @@ const struct kfd2kgd_calls gfx_v12_1_kfd2kgd = {
.set_wave_launch_mode = kgd_gfx_v12_1_set_wave_launch_mode,
.set_address_watch = kgd_gfx_v12_1_set_address_watch,
.clear_address_watch = kgd_gfx_v12_1_clear_address_watch,
- .hqd_sdma_get_doorbell = kgd_gfx_v12_1_hqd_sdma_get_doorbell
+ .hqd_sdma_get_doorbell = kgd_gfx_v12_1_hqd_sdma_get_doorbell,
+ .get_cu_occupancy = kgd_gfx_v12_1_get_cu_occupancy
};
--
2.51.1
^ permalink raw reply related [flat|nested] 2+ messages in thread
* RE: [PATCH] drm/kfd: Add CU occupancy support to GFX12.1
2026-08-17 21:49 [PATCH] drm/kfd: Add CU occupancy support to GFX12.1 David Belanger
@ 2026-08-18 2:50 ` Somasekharan, Sreekant
0 siblings, 0 replies; 2+ messages in thread
From: Somasekharan, Sreekant @ 2026-08-18 2:50 UTC (permalink / raw)
To: Belanger, David, amd-gfx@lists.freedesktop.org; +Cc: Belanger, David
AMD General
The only nitpicks are stale GFX9 carryover comments (VMID vs. doorbell). With those addressed, this patch is
Reviewed-by: Sreekant Somasekharan <Sreekant.Somasekharan@amd.com>
Regards,
-Sreekant
-----Original Message-----
From: amd-gfx <amd-gfx-bounces@lists.freedesktop.org> On Behalf Of David Belanger
Sent: August 17, 2026 5:49 PM
To: amd-gfx@lists.freedesktop.org
Cc: Belanger, David <David.Belanger@amd.com>
Subject: [PATCH] drm/kfd: Add CU occupancy support to GFX12.1
Port changes from GFX9 to GFX12.1 mostly as-is.
Minor changes to register access code.
Assisted-by: Claude:Sonnet 4.6
Signed-off-by: David Belanger <david.belanger@amd.com>
---
.../drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c | 150 +++++++++++++++++-
1 file changed, 149 insertions(+), 1 deletion(-)
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c
index 38ca1aea33b2f..b9a4a365e58ab 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c
@@ -371,6 +371,153 @@ static uint32_t kgd_gfx_v12_1_hqd_sdma_get_doorbell(struct amdgpu_device *adev,
return 0;
}
+static void lock_spi_csq_mutexes(struct amdgpu_device *adev) {
+ mutex_lock(&adev->srbm_mutex);
+ mutex_lock(&adev->grbm_idx_mutex);
+
+}
+
+static void unlock_spi_csq_mutexes(struct amdgpu_device *adev) {
+ mutex_unlock(&adev->grbm_idx_mutex);
+ mutex_unlock(&adev->srbm_mutex);
+}
+
+/**
+ * get_wave_count: Read device registers to get number of waves in
+flight for
+ * a particular queue. The method also returns the doorbell offset
+associated
+ * with the queue.
+ *
+ * @adev: Handle of device whose registers are to be read
+ * @queue_idx: Index of queue in the queue-map bit-field
+ * @queue_cnt: Stores the wave count and doorbell offset for an active
+queue
+ * @inst: xcc's instance number on a multi-XCC setup */ static void
+get_wave_count(struct amdgpu_device *adev, int queue_idx,
+ struct kfd_cu_occupancy *queue_cnt, uint32_t inst) {
+ int pipe_idx;
+ int queue_slot;
+ unsigned int reg_val;
+ unsigned int wave_cnt;
+ /*
+ * Program GRBM with appropriate MEID, PIPEID, QUEUEID and VMID
+ * parameters to read out waves in flight. Get VMID if there are
+ * non-zero waves in flight.
+ */
+ pipe_idx = queue_idx / adev->gfx.mec.num_queue_per_pipe;
+ queue_slot = queue_idx % adev->gfx.mec.num_queue_per_pipe;
+ amdgpu_gfx_select_me_pipe_q(adev, 1, pipe_idx, queue_slot, 0, inst);
+ reg_val = RREG32_SOC15_IP(GC, SOC15_REG_OFFSET(GC, GET_INST(GC, inst),
+ regSPI_CSQ_WF_ACTIVE_COUNT_0) + queue_slot);
+ wave_cnt = reg_val & SPI_CSQ_WF_ACTIVE_COUNT_0__COUNT_MASK;
+ if (wave_cnt != 0) {
+ queue_cnt->wave_cnt += wave_cnt;
+ queue_cnt->doorbell_off =
+ (RREG32_SOC15(GC, GET_INST(GC, inst), regCP_HQD_PQ_DOORBELL_CONTROL) &
+ CP_HQD_PQ_DOORBELL_CONTROL__DOORBELL_OFFSET_MASK) >>
+ CP_HQD_PQ_DOORBELL_CONTROL__DOORBELL_OFFSET__SHIFT;
+ }
+}
+
+/**
+ * kgd_gfx_v12_1_get_cu_occupancy: Reads relevant registers associated
+with
+ * each shader engine and aggregates the number of waves that are in
+flight
+ * for the process whose pasid is provided as a parameter. The process
+could
+ * have ZERO or more queues running and submitting waves to compute units.
+ *
+ * @adev: Handle of device from which to get number of waves in flight
+ * @cu_occupancy: Array that gets filled with wave_cnt and doorbell offset
+ * for comparison later.
+ * @max_waves_per_cu: Output parameter updated with maximum number of waves
+ * possible per Compute Unit
+ * @inst: xcc's instance number on a multi-XCC setup
+ *
+ * Note: It's possible that the device has too many queues
+(oversubscription)
+ * in which case a VMID could be remapped to a different PASID. This
+could lead
+ * to an inaccurate wave count. Following is a high-level sequence:
+ * Time T1: vmid = getVmid(); vmid is associated with Pasid P1
+ * Time T2: passId = getPasId(vmid); vmid is associated with Pasid P2
+ * In the sequence above wave count obtained from time T1 will be
+incorrectly
+ * lost or added to total wave count.
+ *
+ * The registers that provide the waves in flight are:
+ *
+ * SPI_CSQ_WF_ACTIVE_STATUS - bit-map of queues per pipe. The bit is
+ON if a
+ * queue is slotted, OFF if there is no queue. A process could have
+ZERO or
+ * more queues slotted and submitting waves to be run on compute
+units. Even
+ * when there is a queue it is possible there could be zero wave
+fronts, this
+ * can happen when queue is waiting on top-of-pipe events - e.g.
+waitRegMem
+ * command
+ *
+ * For each bit that is ON from above:
+ *
+ * Read (SPI_CSQ_WF_ACTIVE_COUNT_0 + queue_idx) register. It provides the
+ * number of waves that are in flight for the queue at specified index. The
+ * index ranges from 0 to 7.
+ *
+ * If non-zero waves are in flight, store the corresponding doorbell offset
+ * of the queue, along with the wave count.
+ *
+ * Determine if the queue belongs to the process by comparing the doorbell
+ * offset against the process's queues. If it matches, aggregate the wave
+ * count for the process.
+ *
+ * Reading registers referenced above involves programming GRBM
+appropriately */ static void kgd_gfx_v12_1_get_cu_occupancy(struct
+amdgpu_device *adev,
+ struct kfd_cu_occupancy *cu_occupancy,
+ int *max_waves_per_cu, uint32_t inst) {
+ int qidx;
+ int se_idx;
+ int se_cnt;
+ int queue_map;
+ int max_queue_cnt;
+ DECLARE_BITMAP(cp_queue_bitmap, AMDGPU_MAX_QUEUES);
+
+ lock_spi_csq_mutexes(adev);
+ amdgpu_gfx_select_me_pipe_q(adev, 1, 0, 0, 0, inst);
+
+ /*
+ * Iterate through the shader engines and arrays of the device
+ * to get number of waves in flight
+ */
+ bitmap_complement(cp_queue_bitmap, adev->gfx.mec_bitmap[0].queue_bitmap,
+ AMDGPU_MAX_QUEUES);
+ max_queue_cnt = adev->gfx.mec.num_pipe_per_mec *
+ adev->gfx.mec.num_queue_per_pipe;
+ se_cnt = adev->gfx.config.max_shader_engines;
+ for (se_idx = 0; se_idx < se_cnt; se_idx++) {
+ amdgpu_gfx_select_se_sh(adev, se_idx, 0, 0xffffffff, inst);
+ queue_map = RREG32_SOC15(GC, GET_INST(GC, inst),
+ regSPI_CSQ_WF_ACTIVE_STATUS);
+
+ for (qidx = 0; qidx < max_queue_cnt; qidx++) {
+ /* Skip queues that are not associated with
+ * compute functions
+ */
+ if (!test_bit(qidx, cp_queue_bitmap))
+ continue;
+
+ if (!(queue_map & (1 << qidx)))
+ continue;
+
+ /* Get number of waves in flight and aggregate them */
+ get_wave_count(adev, qidx, &cu_occupancy[qidx], inst);
+ }
+ }
+
+ amdgpu_gfx_select_se_sh(adev, 0xffffffff, 0xffffffff, 0xffffffff, inst);
+ amdgpu_gfx_select_me_pipe_q(adev, 0, 0, 0, 0, inst);
+ unlock_spi_csq_mutexes(adev);
+
+ /* Update the output parameters and return */
+ *max_waves_per_cu = adev->gfx.cu_info.simd_per_cu *
+ adev->gfx.cu_info.max_waves_per_simd;
+}
+
const struct kfd2kgd_calls gfx_v12_1_kfd2kgd = {
.init_interrupts = init_interrupts_v12_1,
.hqd_dump = hqd_dump_v12_1,
@@ -384,5 +531,6 @@ const struct kfd2kgd_calls gfx_v12_1_kfd2kgd = {
.set_wave_launch_mode = kgd_gfx_v12_1_set_wave_launch_mode,
.set_address_watch = kgd_gfx_v12_1_set_address_watch,
.clear_address_watch = kgd_gfx_v12_1_clear_address_watch,
- .hqd_sdma_get_doorbell = kgd_gfx_v12_1_hqd_sdma_get_doorbell
+ .hqd_sdma_get_doorbell = kgd_gfx_v12_1_hqd_sdma_get_doorbell,
+ .get_cu_occupancy = kgd_gfx_v12_1_get_cu_occupancy
};
--
2.51.1
^ permalink raw reply related [flat|nested] 2+ messages in thread
end of thread, other threads:[~2026-08-18 2:50 UTC | newest]
Thread overview: 2+ messages (download: mbox.gz follow: Atom feed
-- links below jump to the message on this page --
2026-08-17 21:49 [PATCH] drm/kfd: Add CU occupancy support to GFX12.1 David Belanger
2026-08-18 2:50 ` Somasekharan, Sreekant
This is an external index of several public inboxes,
see mirroring instructions on how to clone and mirror
all data and code used by this external index.