RE: [PATCH v2 2/4] drm/amdgpu/gfx11: recover gfx user queues on priv-fault

"Zhang, Jesse(Jie)" <[email protected]> Thu, 30 Jul 2026 02:12:49 +0000
Newsgroups org.freedesktop.lists.amd-gfx
Message-ID <DM4PR12MB5152DC2113B6BE3D7B294BCEE3C92@DM4PR12MB5152.namprd12.prod.outlook.com>
AMD General

> -----Original Message-----
> From: Alex Deucher <[email protected]>
> Sent: Wednesday, July 29, 2026 10:54 PM
> To: Zhang, Jesse(Jie) <[email protected]>
> Cc: [email protected]; Deucher, Alexander
> <[email protected]>; Koenig, Christian
> <[email protected]>; Mario Sopena-Novales
> <[email protected]>
> Subject: Re: [PATCH v2 2/4] drm/amdgpu/gfx11: recover gfx user queues on priv-
> fault
>
> On Tue, Jul 21, 2026 at 11:39 PM Jesse Zhang <[email protected]> wrote:
> >
> > If a priv/bad-op fault does not match a kernel queue slot, it belongs
> > to a MES-scheduled user queue. Compute IVs carry the doorbell and use
> > the existing path; gfx IVs do not, so record the faulted slot and let
> > a worker read the doorbell back from the HQD
> > (regCP_RB_DOORBELL_CONTROL), look up the user queue and reset it.
> >
> > Suggested-by: Mario Sopena-Novales <[email protected]>
> > Signed-off-by: Jesse Zhang <[email protected]>
> > ---
> >  drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c | 60
> > +++++++++++++++++++++++---
> >  1 file changed, 55 insertions(+), 5 deletions(-)
> >
> > diff --git a/drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c
> > b/drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c
> > index 0cbbdc3694f4..a802b0ac1544 100644
> > --- a/drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c
> > +++ b/drivers/gpu/drm/amd/amdgpu/gfx_v11_0.c
> > @@ -1601,6 +1601,8 @@ static void gfx_v11_0_alloc_ip_dump(struct
> amdgpu_device *adev)
> >         }
> >  }
> >
> > +static void gfx_v11_0_userq_priv_fault_work(struct work_struct
> > +*work);
> > +
> >  static int gfx_v11_0_sw_init(struct amdgpu_ip_block *ip_block)  {
> >         int i, j, k, r, ring_id;
> > @@ -1931,6 +1933,8 @@ static int gfx_v11_0_sw_init(struct
> > amdgpu_ip_block *ip_block)
> >
> >         mutex_init(&adev->gfx.mec.reset_mutex);
> >
> > +       INIT_WORK(&adev->gfx.userq_priv_fault_work,
> > + gfx_v11_0_userq_priv_fault_work);
> > +
> >         return 0;
> >  }
> >
> > @@ -1968,6 +1972,8 @@ static int gfx_v11_0_sw_fini(struct amdgpu_ip_block
> *ip_block)
> >         int i;
> >         struct amdgpu_device *adev = ip_block->adev;
> >
> > +       cancel_work_sync(&adev->gfx.userq_priv_fault_work);
> > +
> >         for (i = 0; i < adev->gfx.num_gfx_rings; i++)
> >                 amdgpu_ring_fini(&adev->gfx.gfx_ring[i]);
> >         for (i = 0; i < adev->gfx.num_compute_rings; i++) @@ -6707,10
> > +6713,47 @@ static int gfx_v11_0_set_priv_inst_fault_state(struct
> amdgpu_device *adev,
> >         return 0;
> >  }
> >
> > +/*
> > + * A gfx exception IV carries no doorbell. For each faulted slot,
> > +read the
> > + * doorbell back from the HQD (left in place by the fatal fault),
> > +look up the
> > + * user queue and kick its per-queue reset.
> > + */
> > +static void gfx_v11_0_userq_priv_fault_work(struct work_struct *work)
> > +{
> > +       struct amdgpu_device *adev =
> > +               container_of(work, struct amdgpu_device, gfx.userq_priv_fault_work);
> > +       unsigned long slots = xchg(&adev->gfx.userq_priv_fault_slots, 0);
> > +       unsigned int id;
> > +
> > +       for_each_set_bit(id, &slots, BITS_PER_LONG) {
> > +               u8 pipe = id & 0x3;
> > +               u8 queue = (id >> 2) & 0x7;
> > +               struct amdgpu_usermode_queue *q;
> > +               u32 db_ctrl, doorbell;
> > +
> > +               amdgpu_gfx_off_ctrl(adev, false);
> > +               mutex_lock(&adev->srbm_mutex);
> > +               soc21_grbm_select(adev, 0, pipe, queue, 0);
> > +               db_ctrl = RREG32_SOC15(GC, 0,
> regCP_RB_DOORBELL_CONTROL);
> > +               soc21_grbm_select(adev, 0, 0, 0, 0);
> > +               mutex_unlock(&adev->srbm_mutex);
> > +               amdgpu_gfx_off_ctrl(adev, true);
> > +
> > +               doorbell = (db_ctrl &
> CP_RB_DOORBELL_CONTROL__DOORBELL_OFFSET_MASK) >>
> > +
> CP_RB_DOORBELL_CONTROL__DOORBELL_OFFSET__SHIFT;
> > +               q = xa_load(&adev->userq_doorbell_xa, doorbell);
> > +               if (q)
> > +                       amdgpu_userq_start_hang_detect_work(q);
> > +       }
> > +}
> > +
> >  static void gfx_v11_0_handle_priv_fault(struct amdgpu_device *adev,
> >                                         struct amdgpu_iv_entry *entry)
> > {
> >         u32 doorbell_offset = entry->src_data[0] &
> > AMDGPU_CTXID0_DOORBELL_ID_MASK;
> > +       u8 me_id = (entry->ring_id & 0x0c) >> 2;
> > +       u8 pipe_id = (entry->ring_id & 0x03) >> 0;
> > +       u8 queue_id = (entry->ring_id & 0x70) >> 4;
> >
> >         /*
> >          * Try KQ first by ring_id (HW slot is authoritative). The @@
> > -6718,9 +6761,6 @@ static void gfx_v11_0_handle_priv_fault(struct
> amdgpu_device *adev,
> >          * never share a HW slot.
> >          */
> >         if (!adev->gfx.disable_kq) {
> > -               u8 me_id = (entry->ring_id & 0x0c) >> 2;
> > -               u8 pipe_id = (entry->ring_id & 0x03) >> 0;
> > -               u8 queue_id = (entry->ring_id & 0x70) >> 4;
> >                 struct amdgpu_ring *ring;
> >                 int i;
> >
> > @@ -6752,10 +6792,20 @@ static void gfx_v11_0_handle_priv_fault(struct
> amdgpu_device *adev,
> >                 }
> >         }
> >
> > -       /* No KQ matched: HW slot is a MES-scheduled user queue. */
> > -       if (adev->enable_mes && doorbell_offset)
> > +       /* No KQ matched: the faulting slot belongs to a user queue. */
> > +       if (!adev->enable_mes)
>
> I think this should be:
>
> if (adev->gfx.disable_uq)
>     return;
>
> Same comment for patch 3.
Thanks Alex, will fix it.
>
> > +               return;
> > +
> > +       if (doorbell_offset) {
> >                 amdgpu_userq_process_reset_irq(adev, entry->pasid,
> >                                                doorbell_offset);
>
> If the doorbell offset is not provided, do we need to keep this code?
> Same comment for patch 3.
both paths are needed. A compute user-queue fault IV carries the doorbell offset, so we reset the queue directly from it.
A gfx user-queue fault is raised by the ME and carries only the HW slot (no doorbell) .-that's exactly why the else branch records the slot and lets the worker read the doorbell back from the HQD.

Thanks
Jesse
>
> Alex
>
> > +       } else {
> > +               /* GFX IVs lack the doorbell; record the slot and let the
> > +                * worker read it back from the HQD.
> > +                */
> > +               set_bit(pipe_id | (queue_id << 2), &adev->gfx.userq_priv_fault_slots);
> > +               schedule_work(&adev->gfx.userq_priv_fault_work);
> > +       }
> >  }
> >
> >  static int gfx_v11_0_priv_reg_irq(struct amdgpu_device *adev,
> > --
> > 2.49.0
> >