[PATCH V15 11/14] drm/xe/vram: Use RCU for lock-free sysfs reads of bad page lists

Tejas Upadhyay <[email protected]>
Newsgroups org.freedesktop.lists.intel-xe
Message-ID <[email protected]>
The sysfs vram_bad_pages reader previously held mgr->lock while
formatting the entire output, blocking normal VRAM alloc/free
operations for the duration of the read.

Switch to RCU-protected list traversal for the sysfs read path:

Writer side (page offline, under mgr->lock):
  - list_add()  -> list_add_rcu()
  - list_del()  -> list_del_rcu()
  - kfree()     -> kfree_rcu()

Reader side (sysfs serialize_bad_pages):
  - Drop mgr->lock entirely
  - Use rcu_read_lock() + list_for_each_entry_rcu()
  - Use READ_ONCE() for entry counters

The writer-side xe_ttm_vram_page_already_processed() keeps
lockdep_assert_held(&mgr->lock) since it requires serialization
against concurrent page offline operations.

Signed-off-by: Tejas Upadhyay <[email protected]>
---
 drivers/gpu/drm/xe/xe_ttm_vram_mgr.c       | 161 +++++++++++++++++++--
 drivers/gpu/drm/xe/xe_ttm_vram_mgr_types.h |   4 +
 2 files changed, 155 insertions(+), 10 deletions(-)

diff --git a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
index 6280886e2ebb..c22669955147 100644
--- a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
+++ b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c
@@ -312,15 +312,15 @@ static void xe_ttm_vram_free_bad_pages(struct drm_device *dev, struct xe_ttm_vra
 
 	list_for_each_entry_safe(pos, n, &mgr->offlined_pages, offlined_link) {
 		xe_ttm_vram_buddy_free(mgr, &pos->blocks, pos->used_visible_size);
-		list_del(&pos->offlined_link);
+		list_del_rcu(&pos->offlined_link);
 		--mgr->n_offlined_pages;
-		kfree(pos);
+		kfree_rcu(pos, rcu);
 	}
 	list_for_each_entry_safe(pos, n, &mgr->queued_pages, queued_link) {
 		xe_ttm_vram_buddy_free(mgr, &pos->blocks, 0);
-		list_del(&pos->queued_link);
+		list_del_rcu(&pos->queued_link);
 		--mgr->n_queued_pages;
-		kfree(pos);
+		kfree_rcu(pos, rcu);
 	}
 }
 
@@ -657,7 +657,7 @@ static int xe_ttm_vram_reserve_page_at_addr(struct xe_device *xe, u64 addr,
 				break;
 			}
 			++vram_mgr->n_queued_pages;
-			list_add(&nentry->queued_link, &vram_mgr->queued_pages);
+			list_add_rcu(&nentry->queued_link, &vram_mgr->queued_pages);
 		}
 	}
 
@@ -702,11 +702,11 @@ static int xe_ttm_vram_reserve_page_at_addr(struct xe_device *xe, u64 addr,
 			list_for_each_entry_safe(pos, n, &vram_mgr->queued_pages, queued_link) {
 				if (pos->addr == nentry->addr) {
 					--vram_mgr->n_queued_pages;
-					list_del(&pos->queued_link);
+					list_del_rcu(&pos->queued_link);
 					break;
 				}
 			}
-			list_add(&nentry->offlined_link, &vram_mgr->offlined_pages);
+			list_add_rcu(&nentry->offlined_link, &vram_mgr->offlined_pages);
 			/* RAS will send command to FW for offlining page based on ret value */
 			++vram_mgr->n_offlined_pages;
 			return ret;
@@ -716,7 +716,7 @@ static int xe_ttm_vram_reserve_page_at_addr(struct xe_device *xe, u64 addr,
 
 		scoped_guard(mutex, &vram_mgr->lock) {
 			++vram_mgr->n_queued_pages;
-			list_add(&nentry->queued_link, &vram_mgr->queued_pages);
+			list_add_rcu(&nentry->queued_link, &vram_mgr->queued_pages);
 			ret = xe_ttm_vram_buddy_alloc(vram_mgr, addr, addr + size,
 						      size, size, &nentry->blocks,
 						      GPU_BUDDY_RANGE_ALLOCATION,
@@ -732,12 +732,12 @@ static int xe_ttm_vram_reserve_page_at_addr(struct xe_device *xe, u64 addr,
 			list_for_each_entry_safe(pos, n, &vram_mgr->queued_pages, queued_link) {
 				if (pos->addr == nentry->addr) {
 					--vram_mgr->n_queued_pages;
-					list_del(&pos->queued_link);
+					list_del_rcu(&pos->queued_link);
 					break;
 				}
 			}
 			++vram_mgr->n_offlined_pages;
-			list_add(&nentry->offlined_link, &vram_mgr->offlined_pages);
+			list_add_rcu(&nentry->offlined_link, &vram_mgr->offlined_pages);
 			/* RAS will send command to FW for offlining page based on ret value */
 		}
 	}
@@ -825,3 +825,144 @@ int xe_ttm_vram_handle_addr_fault(struct xe_device *xe, u64 addr)
 	return xe_ttm_vram_reserve_page_at_addr(xe, addr, vram_mgr, mm);
 }
 EXPORT_SYMBOL(xe_ttm_vram_handle_addr_fault);
+
+static size_t serialize_bad_pages(struct xe_ttm_vram_mgr *mgr, char *buf, size_t max_len)
+{
+	struct xe_ttm_vram_offline_resource *pos;
+	struct gpu_buddy_block *block;
+	size_t s = 0;
+	int printed;
+	int count = 0;
+
+	rcu_read_lock();
+
+	printed = scnprintf(buf + s, max_len - s, "max_pages: %d\n", mgr->max_pages);
+	s += printed;
+
+	list_for_each_entry_rcu(pos, &mgr->offlined_pages, offlined_link) {
+		if (count >= 10000 || s >= max_len)
+			break;
+
+		block = list_first_entry_or_null(&pos->blocks, struct gpu_buddy_block, link);
+		if (!block)
+			continue;
+
+		printed = scnprintf(buf + s, max_len - s, "0x%016llx : 0x%016llx : %c\n",
+				    gpu_buddy_block_offset(block) >> PAGE_SHIFT,
+				    gpu_buddy_block_size(&mgr->mm, block), 'R');
+		s += printed;
+		count++;
+	}
+	list_for_each_entry_rcu(pos, &mgr->queued_pages, queued_link) {
+		u64 pfn, blk_size;
+
+		if (count >= 10000 || s >= max_len)
+			break;
+
+		block = list_first_entry_or_null(&pos->blocks, struct gpu_buddy_block, link);
+		if (block) {
+			pfn = gpu_buddy_block_offset(block) >> PAGE_SHIFT;
+			blk_size = gpu_buddy_block_size(&mgr->mm, block);
+		} else {
+			pfn = pos->addr >> PAGE_SHIFT;
+			blk_size = PAGE_SIZE;
+		}
+
+		printed = scnprintf(buf + s, max_len - s, "0x%016llx : 0x%016llx : %c\n",
+				    pfn, blk_size, pos->status ? 'F' : 'P');
+		s += printed;
+		count++;
+	}
+
+	rcu_read_unlock();
+	return s;
+}
+
+static ssize_t vram_bad_pages_bin_read(struct file *filp, struct kobject *kobj,
+				       const struct bin_attribute *attr, char *buf,
+				       loff_t off, size_t count)
+{
+	struct device *dev = kobj_to_dev(kobj);
+	struct pci_dev *pdev = to_pci_dev(dev);
+	struct ttm_resource_manager *man;
+	struct xe_ttm_vram_mgr *mgr;
+	size_t allocation_size;
+	struct xe_device *xe;
+	size_t full_data_len;
+	int active_entries;
+	char *temp_buf;
+
+	xe = pdev_to_xe_device(pdev);
+	man = ttm_manager_type(&xe->ttm, XE_PL_VRAM0);
+	if (!man)
+		return -ENODEV;
+	mgr = to_xe_ttm_vram_mgr(man);
+
+	active_entries = READ_ONCE(mgr->n_offlined_pages) + READ_ONCE(mgr->n_queued_pages);
+
+	if (active_entries > 10000)
+		active_entries = 10000;
+
+	allocation_size = 64 + (active_entries * 48);
+
+	temp_buf = kvmalloc(allocation_size, GFP_KERNEL);
+	if (!temp_buf)
+		return -ENOMEM;
+
+	/* serialize_bad_pages uses rcu_read_lock internally */
+	full_data_len = serialize_bad_pages(mgr, temp_buf, allocation_size);
+
+	if (off >= full_data_len) {
+		kvfree(temp_buf);
+		return 0;
+	}
+
+	if (off + count > full_data_len)
+		count = full_data_len - off;
+
+	memcpy(buf, temp_buf + off, count);
+
+	kvfree(temp_buf);
+	return count;
+}
+
+static const struct bin_attribute bin_attr_vram_bad_pages = {
+	.attr = { .name = "vram_bad_pages", .mode = 0444 },
+	.read = vram_bad_pages_bin_read,
+	.size = 0,
+};
+
+static void xe_ttm_vram_sysfs_fini(void *arg)
+{
+	struct xe_device *xe = arg;
+	struct pci_dev *pdev = to_pci_dev(xe->drm.dev);
+
+	sysfs_remove_bin_file(&pdev->dev.kobj, &bin_attr_vram_bad_pages);
+}
+
+/**
+ * xe_ttm_vram_sysfs_init - Initialize vram bad pages sysfs binary file
+ * @xe: Xe Device object
+ *
+ * Creates a binary sysfs file under the PCI device for reading
+ * offlined and queued VRAM pages. Supports large entry counts
+ * via offset/count pagination.
+ *
+ * Returns: 0 on success, negative error code on error.
+ */
+int xe_ttm_vram_sysfs_init(struct xe_device *xe)
+{
+	struct pci_dev *pdev = to_pci_dev(xe->drm.dev);
+	int err;
+
+	err = sysfs_create_bin_file(&pdev->dev.kobj, &bin_attr_vram_bad_pages);
+	if (err) {
+		dev_err(&pdev->dev,
+			"Failed to create vram_bad_pages sysfs: %d\n",
+			err);
+		return err;
+	}
+
+	return devm_add_action_or_reset(&pdev->dev, xe_ttm_vram_sysfs_fini, xe);
+}
+EXPORT_SYMBOL(xe_ttm_vram_sysfs_init);
diff --git a/drivers/gpu/drm/xe/xe_ttm_vram_mgr_types.h b/drivers/gpu/drm/xe/xe_ttm_vram_mgr_types.h
index bdfdf6ec1218..003d3a7cb1dd 100644
--- a/drivers/gpu/drm/xe/xe_ttm_vram_mgr_types.h
+++ b/drivers/gpu/drm/xe/xe_ttm_vram_mgr_types.h
@@ -37,6 +37,8 @@ struct xe_ttm_vram_mgr {
 	struct mutex lock;
 	/** @mem_type: The TTM memory type */
 	u32 mem_type;
+	/** @max_pages: max pages that can be in offline queue retrieved from FW */
+	u16 max_pages;
 };
 
 /**
@@ -69,6 +71,8 @@ struct xe_ttm_vram_offline_resource {
 	u64 addr;
 	/** @status: Reservation status (0=pending, 1=fail) */
 	bool status;
+	/** @rcu: RCU head for deferred freeing */
+	struct rcu_head rcu;
 };
 
 #endif
-- 
2.52.0
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.