RE: [PATCH] drm/kfd: Add CU occupancy support to GFX12.1

"Somasekharan, Sreekant" <[email protected]>
Newsgroups org.freedesktop.lists.amd-gfx
Message-ID <MN6PR12MB8590CC24098118FF3C86E75CFBA62@MN6PR12MB8590.namprd12.prod.outlook.com>
AMD General

The only nitpicks are stale GFX9 carryover comments (VMID vs. doorbell). With those addressed, this patch is

Reviewed-by: Sreekant Somasekharan <[email protected]>

Regards,

-Sreekant


-----Original Message-----
From: amd-gfx <[email protected]> On Behalf Of David Belanger
Sent: August 17, 2026 5:49 PM
To: [email protected]
Cc: Belanger, David <[email protected]>
Subject: [PATCH] drm/kfd: Add CU occupancy support to GFX12.1

Port changes from GFX9 to GFX12.1 mostly as-is.
Minor changes to register access code.

Assisted-by: Claude:Sonnet 4.6
Signed-off-by: David Belanger <[email protected]>
---
 .../drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c  | 150 +++++++++++++++++-
 1 file changed, 149 insertions(+), 1 deletion(-)

diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c
index 38ca1aea33b2f..b9a4a365e58ab 100644
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gfx_v12_1.c
@@ -371,6 +371,153 @@ static uint32_t kgd_gfx_v12_1_hqd_sdma_get_doorbell(struct amdgpu_device *adev,
        return 0;
 }

+static void lock_spi_csq_mutexes(struct amdgpu_device *adev) {
+       mutex_lock(&adev->srbm_mutex);
+       mutex_lock(&adev->grbm_idx_mutex);
+
+}
+
+static void unlock_spi_csq_mutexes(struct amdgpu_device *adev) {
+       mutex_unlock(&adev->grbm_idx_mutex);
+       mutex_unlock(&adev->srbm_mutex);
+}
+
+/**
+ * get_wave_count: Read device registers to get number of waves in
+flight for
+ * a particular queue. The method also returns the doorbell offset
+associated
+ * with the queue.
+ *
+ * @adev: Handle of device whose registers are to be read
+ * @queue_idx: Index of queue in the queue-map bit-field
+ * @queue_cnt: Stores the wave count and doorbell offset for an active
+queue
+ * @inst: xcc's instance number on a multi-XCC setup  */ static void
+get_wave_count(struct amdgpu_device *adev, int queue_idx,
+               struct kfd_cu_occupancy *queue_cnt, uint32_t inst) {
+       int pipe_idx;
+       int queue_slot;
+       unsigned int reg_val;
+       unsigned int wave_cnt;
+       /*
+        * Program GRBM with appropriate MEID, PIPEID, QUEUEID and VMID
+        * parameters to read out waves in flight. Get VMID if there are
+        * non-zero waves in flight.
+        */
+       pipe_idx = queue_idx / adev->gfx.mec.num_queue_per_pipe;
+       queue_slot = queue_idx % adev->gfx.mec.num_queue_per_pipe;
+       amdgpu_gfx_select_me_pipe_q(adev, 1, pipe_idx, queue_slot, 0, inst);
+       reg_val = RREG32_SOC15_IP(GC, SOC15_REG_OFFSET(GC, GET_INST(GC, inst),
+                                 regSPI_CSQ_WF_ACTIVE_COUNT_0) + queue_slot);
+       wave_cnt = reg_val & SPI_CSQ_WF_ACTIVE_COUNT_0__COUNT_MASK;
+       if (wave_cnt != 0) {
+               queue_cnt->wave_cnt += wave_cnt;
+               queue_cnt->doorbell_off =
+                       (RREG32_SOC15(GC, GET_INST(GC, inst), regCP_HQD_PQ_DOORBELL_CONTROL) &
+                        CP_HQD_PQ_DOORBELL_CONTROL__DOORBELL_OFFSET_MASK) >>
+                        CP_HQD_PQ_DOORBELL_CONTROL__DOORBELL_OFFSET__SHIFT;
+       }
+}
+
+/**
+ * kgd_gfx_v12_1_get_cu_occupancy: Reads relevant registers associated
+with
+ * each shader engine and aggregates the number of waves that are in
+flight
+ * for the process whose pasid is provided as a parameter. The process
+could
+ * have ZERO or more queues running and submitting waves to compute units.
+ *
+ * @adev: Handle of device from which to get number of waves in flight
+ * @cu_occupancy: Array that gets filled with wave_cnt and doorbell offset
+ *               for comparison later.
+ * @max_waves_per_cu: Output parameter updated with maximum number of waves
+ *                    possible per Compute Unit
+ * @inst: xcc's instance number on a multi-XCC setup
+ *
+ * Note: It's possible that the device has too many queues
+(oversubscription)
+ * in which case a VMID could be remapped to a different PASID. This
+could lead
+ * to an inaccurate wave count. Following is a high-level sequence:
+ *    Time T1: vmid = getVmid(); vmid is associated with Pasid P1
+ *    Time T2: passId = getPasId(vmid); vmid is associated with Pasid P2
+ * In the sequence above wave count obtained from time T1 will be
+incorrectly
+ * lost or added to total wave count.
+ *
+ * The registers that provide the waves in flight are:
+ *
+ *  SPI_CSQ_WF_ACTIVE_STATUS - bit-map of queues per pipe. The bit is
+ON if a
+ *  queue is slotted, OFF if there is no queue. A process could have
+ZERO or
+ *  more queues slotted and submitting waves to be run on compute
+units. Even
+ *  when there is a queue it is possible there could be zero wave
+fronts, this
+ *  can happen when queue is waiting on top-of-pipe events - e.g.
+waitRegMem
+ *  command
+ *
+ *  For each bit that is ON from above:
+ *
+ *    Read (SPI_CSQ_WF_ACTIVE_COUNT_0 + queue_idx) register. It provides the
+ *    number of waves that are in flight for the queue at specified index. The
+ *    index ranges from 0 to 7.
+ *
+ *    If non-zero waves are in flight, store the corresponding doorbell offset
+ *    of the queue, along with the wave count.
+ *
+ *    Determine if the queue belongs to the process by comparing the doorbell
+ *    offset against the process's queues. If it matches, aggregate the wave
+ *    count for the process.
+ *
+ *  Reading registers referenced above involves programming GRBM
+appropriately  */ static void kgd_gfx_v12_1_get_cu_occupancy(struct
+amdgpu_device *adev,
+                               struct kfd_cu_occupancy *cu_occupancy,
+                               int *max_waves_per_cu, uint32_t inst) {
+       int qidx;
+       int se_idx;
+       int se_cnt;
+       int queue_map;
+       int max_queue_cnt;
+       DECLARE_BITMAP(cp_queue_bitmap, AMDGPU_MAX_QUEUES);
+
+       lock_spi_csq_mutexes(adev);
+       amdgpu_gfx_select_me_pipe_q(adev, 1, 0, 0, 0, inst);
+
+       /*
+        * Iterate through the shader engines and arrays of the device
+        * to get number of waves in flight
+        */
+       bitmap_complement(cp_queue_bitmap, adev->gfx.mec_bitmap[0].queue_bitmap,
+                         AMDGPU_MAX_QUEUES);
+       max_queue_cnt = adev->gfx.mec.num_pipe_per_mec *
+                       adev->gfx.mec.num_queue_per_pipe;
+       se_cnt = adev->gfx.config.max_shader_engines;
+       for (se_idx = 0; se_idx < se_cnt; se_idx++) {
+               amdgpu_gfx_select_se_sh(adev, se_idx, 0, 0xffffffff, inst);
+               queue_map = RREG32_SOC15(GC, GET_INST(GC, inst),
+                                        regSPI_CSQ_WF_ACTIVE_STATUS);
+
+               for (qidx = 0; qidx < max_queue_cnt; qidx++) {
+                       /* Skip queues that are not associated with
+                        * compute functions
+                        */
+                       if (!test_bit(qidx, cp_queue_bitmap))
+                               continue;
+
+                       if (!(queue_map & (1 << qidx)))
+                               continue;
+
+                       /* Get number of waves in flight and aggregate them */
+                       get_wave_count(adev, qidx, &cu_occupancy[qidx], inst);
+               }
+       }
+
+       amdgpu_gfx_select_se_sh(adev, 0xffffffff, 0xffffffff, 0xffffffff, inst);
+       amdgpu_gfx_select_me_pipe_q(adev, 0, 0, 0, 0, inst);
+       unlock_spi_csq_mutexes(adev);
+
+       /* Update the output parameters and return */
+       *max_waves_per_cu = adev->gfx.cu_info.simd_per_cu *
+                               adev->gfx.cu_info.max_waves_per_simd;
+}
+
 const struct kfd2kgd_calls gfx_v12_1_kfd2kgd = {
        .init_interrupts = init_interrupts_v12_1,
        .hqd_dump = hqd_dump_v12_1,
@@ -384,5 +531,6 @@ const struct kfd2kgd_calls gfx_v12_1_kfd2kgd = {
        .set_wave_launch_mode = kgd_gfx_v12_1_set_wave_launch_mode,
        .set_address_watch = kgd_gfx_v12_1_set_address_watch,
        .clear_address_watch = kgd_gfx_v12_1_clear_address_watch,
-       .hqd_sdma_get_doorbell = kgd_gfx_v12_1_hqd_sdma_get_doorbell
+       .hqd_sdma_get_doorbell = kgd_gfx_v12_1_hqd_sdma_get_doorbell,
+       .get_cu_occupancy = kgd_gfx_v12_1_get_cu_occupancy
 };
--
2.51.1
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.