[PATCH i-g-t 2/2] tests/amdgpu/amd_deadlock: add gfx user-queue priv-fault reset test

Jesse Zhang <[email protected]>
Newsgroups org.freedesktop.lists.igt-dev
Message-ID <[email protected]>
Submit a gfx user-queue job that raises a priv-fault interrupt and then
hangs cleanly: a minimal invalid opcode (0xf2) makes the CP raise
CP_BAD_OPCODE_ERROR without running off into memory, and a trailing
WAIT_REG_MEM that never completes keeps the queue hung. The driver recovers
it with a per-queue reset instead of a full GPU reset. The job is submitted
on the synchronised path so it blocks until the reset completes the fence.
Add it as the amdgpu-gfx-priv-fault-umq subtest, gated on
AMDGPU_ENABLE_USERQTEST and per-queue reset capability.

Signed-off-by: Jesse Zhang <[email protected]>
---
 lib/amdgpu/amd_deadlock_helpers.c | 108 ++++++++++++++++++++++++++++++
 lib/amdgpu/amd_deadlock_helpers.h |   4 ++
 tests/amdgpu/amd_deadlock.c       |   9 +++
 3 files changed, 121 insertions(+)

diff --git a/lib/amdgpu/amd_deadlock_helpers.c b/lib/amdgpu/amd_deadlock_helpers.c
index c679e60c7..402972b5c 100644
--- a/lib/amdgpu/amd_deadlock_helpers.c
+++ b/lib/amdgpu/amd_deadlock_helpers.c
@@ -503,6 +503,114 @@ void amdgpu_hang_ring_helper(amdgpu_device_handle device_handle, unsigned int ip
 	}
 }
 
+/*
+ * Build a packet stream that raises a gfx priv-fault interrupt and then hangs
+ * the queue cleanly: a minimal invalid opcode (CP_BAD_OPCODE_ERROR) followed by
+ * a WAIT_REG_MEM that never completes. The bad opcode alone would let the CP
+ * run to completion (self-recovering), and a bad opcode with a mis-parseable
+ * body drags the CP into a page fault (only recoverable by a full GPU reset).
+ * Combining a minimal bad opcode with a clean wait gives the wanted case: the
+ * priv-fault interrupt fires, the queue hangs without faulting the CP, and the
+ * driver recovers it with a per-queue reset.
+ */
+static void gfx_ring_emit_priv_fault_hang(struct amdgpu_ring_context *ring_context)
+{
+	uint32_t i = 0;
+
+	/* Invalid opcode: the CP fails to decode it and raises a
+	 * CP_BAD_OPCODE_ERROR interrupt (a gfx priv-fault). Keep it minimal so
+	 * the CP does not run off into memory.
+	 */
+	ring_context->pm4[i++] = PACKET3(0xf2, 0);
+	ring_context->pm4[i++] = 0x0;
+
+	/* Then wait forever on a resident BO (initialised to 0, polled for
+	 * != 0). The queue hangs cleanly with no page fault, so the driver can
+	 * recover it with a per-queue reset instead of a full GPU reset.
+	 */
+	ring_context->pm4[i++] = PACKET3(PACKET3_WAIT_REG_MEM, 5);
+	ring_context->pm4[i++] = WAIT_REG_MEM_MEM_SPACE(1) |
+				 WAIT_REG_MEM_FUNCTION(4) |
+				 WAIT_REG_MEM_ENGINE(0);
+	ring_context->pm4[i++] = ring_context->bo_mc & 0xfffffffc;
+	ring_context->pm4[i++] = (ring_context->bo_mc >> 32) & 0xffffffff;
+	ring_context->pm4[i++] = 0;          /* reference value */
+	ring_context->pm4[i++] = 0xffffffff; /* and mask */
+	ring_context->pm4[i++] = 0x00000004; /* poll interval */
+	ring_context->pm4_dw = i;
+}
+
+/*
+ * Fault a user queue with an invalid opcode followed by an endless wait: the
+ * bad opcode raises the gfx priv-fault interrupt and the wait hangs the queue
+ * cleanly, so the driver recovers it with a per-queue reset (no full GPU
+ * reset). The faulting submit uses the normal (synchronised) path so it blocks
+ * until the per-queue reset completes the fence.
+ */
+void amdgpu_priv_fault_ring_helper(amdgpu_device_handle device_handle, unsigned int ip_type,
+				   struct pci_addr *pci, bool user_queue)
+{
+	const struct amdgpu_ip_block_version *ip_block;
+	const int write_length = 128;
+	const int pm4_dw = 256;
+	struct amdgpu_ring_context *ring_context;
+	int r = 0;
+
+	ip_block = get_ip_block(device_handle, ip_type);
+	ring_context = calloc(1, sizeof(*ring_context));
+	igt_assert(ring_context);
+
+	if (user_queue) {
+		ip_block->funcs->userq_create(device_handle, ring_context, ip_type);
+	} else {
+		r = amdgpu_cs_ctx_create(device_handle, &ring_context->context_handle);
+		igt_assert_eq(r, 0);
+	}
+
+	ring_context->write_length = write_length;
+	ring_context->pm4 = calloc(pm4_dw, sizeof(*ring_context->pm4));
+	ring_context->pm4_size = pm4_dw;
+	ring_context->res_cnt = 1;
+	ring_context->ring_id = 0;
+	ring_context->user_queue = user_queue;
+	igt_assert(ring_context->pm4);
+
+	r = amdgpu_bo_alloc_and_map_sync(device_handle,
+				    ring_context->write_length * sizeof(uint32_t),
+				    4096, AMDGPU_GEM_DOMAIN_GTT,
+				    AMDGPU_GEM_CREATE_CPU_GTT_USWC,
+				    AMDGPU_VM_MTYPE_UC,
+				    &ring_context->bo,
+				    (void **)&ring_context->bo_cpu,
+				    &ring_context->bo_mc,
+				    &ring_context->va_handle,
+				    ring_context->timeline_syncobj_handle,
+				    ++ring_context->point, user_queue);
+	igt_assert_eq(r, 0);
+	if (user_queue) {
+		r = amdgpu_timeline_syncobj_wait(device_handle,
+			ring_context->timeline_syncobj_handle,
+			ring_context->point);
+		igt_assert_eq(r, 0);
+	}
+
+	memset((void *)ring_context->bo_cpu, 0, ring_context->write_length * sizeof(uint32_t));
+	ring_context->resources[0] = ring_context->bo;
+
+	gfx_ring_emit_priv_fault_hang(ring_context);
+
+	amdgpu_test_exec_cs_helper(device_handle, ip_block->type, ring_context, 0);
+
+	amdgpu_bo_unmap_and_free(ring_context->bo, ring_context->va_handle, ring_context->bo_mc,
+				 ring_context->write_length * sizeof(uint32_t));
+	if (user_queue) {
+		ip_block->funcs->userq_destroy(device_handle, ring_context, ip_type);
+	} else {
+		free(ring_context->pm4);
+		free(ring_context);
+	}
+}
+
 #define MAX_DMABUF_COUNT 0x20000
 #define MAX_DWORD_COUNT 256
 
diff --git a/lib/amdgpu/amd_deadlock_helpers.h b/lib/amdgpu/amd_deadlock_helpers.h
index accbf2e41..6e69e62cb 100644
--- a/lib/amdgpu/amd_deadlock_helpers.h
+++ b/lib/amdgpu/amd_deadlock_helpers.h
@@ -37,5 +37,9 @@ amdgpu_hang_sdma_ring_helper(amdgpu_device_handle device_handle, uint8_t hang_ty
 void
 amdgpu_hang_ring_helper(amdgpu_device_handle device_handle, unsigned int ip_type,
 			struct pci_addr *pci, bool user_queue);
+
+void
+amdgpu_priv_fault_ring_helper(amdgpu_device_handle device_handle, unsigned int ip_type,
+			      struct pci_addr *pci, bool user_queue);
 #endif
 
diff --git a/tests/amdgpu/amd_deadlock.c b/tests/amdgpu/amd_deadlock.c
index 118948992..0d0e4ba6e 100644
--- a/tests/amdgpu/amd_deadlock.c
+++ b/tests/amdgpu/amd_deadlock.c
@@ -255,6 +255,15 @@ int igt_main()
 			amdgpu_hang_ring_helper(device, AMDGPU_HW_IP_GFX, &pci, true);
 		}
 	}
+
+	igt_describe("Test-per-queue-reset-recovery-of-a-gfx-user-queue-priv-fault");
+	igt_subtest_with_dynamic("amdgpu-gfx-priv-fault-umq") {
+		if (enable_test && userq_arr_cap[AMD_IP_GFX] &&
+			is_reset_enable(AMD_IP_GFX, AMDGPU_RESET_TYPE_PER_QUEUE, &pci)) {
+			igt_dynamic_f("amdgpu-gfx-priv-fault-umq")
+			amdgpu_priv_fault_ring_helper(device, AMDGPU_HW_IP_GFX, &pci, true);
+		}
+	}
 #endif
 
 	igt_fixture() {
-- 
2.49.0
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.