Re: [PATCH V15 10/14] drm/xe/configfs: Add vram bad page reservation policy
"Ghimiray, Himal Prasad" <[email protected]>
| Newsgroups | org.freedesktop.lists.intel-xe |
|---|---|
| Message-ID | <[email protected]> |
On 11-08-2026 18:10, Tejas Upadhyay wrote: > The interface enables setting the policy for how bad pages are > handled in VRAM. This is crucial for maintaining system > stability in scenarios where VRAM degradation occurs. > > By default policy will be "reserve", which can be changed to > "logging" only. > > v3: > - All FW communication moved under RAS > v2: > - Add CRI check and rebase > > Signed-off-by: Tejas Upadhyay <[email protected]> > --- > drivers/gpu/drm/xe/xe_configfs.c | 64 +++++++++++++++++++++++++++- > drivers/gpu/drm/xe/xe_configfs.h | 2 + > drivers/gpu/drm/xe/xe_ttm_vram_mgr.c | 10 +++++ > 3 files changed, 75 insertions(+), 1 deletion(-) > > diff --git a/drivers/gpu/drm/xe/xe_configfs.c b/drivers/gpu/drm/xe/xe_configfs.c > index 052cce962161..c4f386d4bf09 100644 > --- a/drivers/gpu/drm/xe/xe_configfs.c > +++ b/drivers/gpu/drm/xe/xe_configfs.c > @@ -61,7 +61,8 @@ > * ├── survivability_mode > * ├── gt_types_allowed > * ├── engines_allowed > - * └── enable_psmi > + * ├── enable_psmi > + * └── bad_page_reservation > * > * After configuring the attributes as per next section, the device can be > * probed with:: > @@ -159,6 +160,16 @@ > * > * This attribute can only be set before binding to the device. > * > + * Bad pages reservation: > + * --------------------- > + * > + * Disable vram bad pages reservation, instead just report it in dmesg. > + * Example to disable it:: > + * > + * # echo 0 > /sys/kernel/config/xe/0000:03:00.0/bad_page_reservation > + * > + * This attribute can only be set before binding to the device. > + * > * Context restore BB > * ------------------ > * > @@ -275,6 +286,7 @@ struct xe_config_group_device { > bool survivability_mode; > bool enable_psmi; > bool enable_multi_queue; > + bool bad_page_reservation; > struct { > unsigned int max_vfs; > bool admin_only_pf; > @@ -295,6 +307,7 @@ static const struct xe_config_device device_defaults = { > .survivability_mode = false, > .enable_psmi = false, > .enable_multi_queue = true, > + .bad_page_reservation = true, > .sriov = { > .max_vfs = XE_DEFAULT_MAX_VFS, > .admin_only_pf = XE_DEFAULT_ADMIN_ONLY_PF, > @@ -616,6 +629,32 @@ static ssize_t enable_multi_queue_store(struct config_item *item, const char *pa > return len; > } > > +static ssize_t bad_page_reservation_show(struct config_item *item, char *page) > +{ > + struct xe_config_device *dev = to_xe_config_device(item); > + > + return sprintf(page, "%d\n", dev->bad_page_reservation); > +} > + > +static ssize_t bad_page_reservation_store(struct config_item *item, const char *page, size_t len) > +{ > + struct xe_config_group_device *dev = to_xe_config_group_device(item); > + bool val; > + int ret; > + > + ret = kstrtobool(page, &val); > + if (ret) > + return ret; > + > + guard(mutex)(&dev->lock); > + if (is_bound(dev)) > + return -EBUSY; > + > + dev->config.bad_page_reservation = val; > + > + return len; > +} > + > static bool wa_bb_read_advance(bool dereference, char **p, > const char *append, size_t len, > size_t *max_size) > @@ -855,6 +894,7 @@ CONFIGFS_ATTR(, ctx_restore_mid_bb); > CONFIGFS_ATTR(, ctx_restore_post_bb); > CONFIGFS_ATTR(, enable_multi_queue); > CONFIGFS_ATTR(, enable_psmi); > +CONFIGFS_ATTR(, bad_page_reservation); > CONFIGFS_ATTR(, engines_allowed); > CONFIGFS_ATTR(, gt_types_allowed); > CONFIGFS_ATTR(, survivability_mode); > @@ -864,6 +904,7 @@ static struct configfs_attribute *xe_config_device_attrs[] = { > &attr_ctx_restore_post_bb, > &attr_enable_multi_queue, > &attr_enable_psmi, > + &attr_bad_page_reservation, > &attr_engines_allowed, > &attr_gt_types_allowed, > &attr_survivability_mode, > @@ -1142,6 +1183,7 @@ static void dump_custom_dev_config(struct pci_dev *pdev, > PRI_CUSTOM_ATTR("%llx", engines_allowed); > PRI_CUSTOM_ATTR("%d", enable_multi_queue); > PRI_CUSTOM_ATTR("%d", enable_psmi); > + PRI_CUSTOM_ATTR("%d", bad_page_reservation); > PRI_CUSTOM_ATTR("%d", survivability_mode); > PRI_CUSTOM_ATTR("%u", sriov.admin_only_pf); > > @@ -1290,6 +1332,26 @@ bool xe_configfs_get_enable_multi_queue(struct pci_dev *pdev) > return ret; > } > > +/** > + * xe_configfs_get_bad_page_reservation - get configfs bad_page_reservation setting > + * @pdev: pci device > + * > + * Return: bad_page_reservation setting in configfs Nit: Better to document configfs val 0 means Logging only 1 means logging and offlining With that Reviewed-by: Himal Prasad Ghimiray <[email protected]> > + */ > +bool xe_configfs_get_bad_page_reservation(struct pci_dev *pdev) > +{ > + struct xe_config_group_device *dev = find_xe_config_group_device(pdev); > + bool ret; > + > + if (!dev) > + return device_defaults.bad_page_reservation; > + > + ret = dev->config.bad_page_reservation; > + config_group_put(&dev->group); > + > + return ret; > +} > + > /** > * xe_configfs_get_ctx_restore_mid_bb - get configfs ctx_restore_mid_bb setting > * @pdev: pci device > diff --git a/drivers/gpu/drm/xe/xe_configfs.h b/drivers/gpu/drm/xe/xe_configfs.h > index 4fbbeafba473..7405cc5f3207 100644 > --- a/drivers/gpu/drm/xe/xe_configfs.h > +++ b/drivers/gpu/drm/xe/xe_configfs.h > @@ -24,6 +24,7 @@ bool xe_configfs_media_gt_allowed(struct pci_dev *pdev); > u64 xe_configfs_get_engines_allowed(struct pci_dev *pdev); > bool xe_configfs_get_psmi_enabled(struct pci_dev *pdev); > bool xe_configfs_get_enable_multi_queue(struct pci_dev *pdev); > +bool xe_configfs_get_bad_page_reservation(struct pci_dev *pdev); > u32 xe_configfs_get_ctx_restore_mid_bb(struct pci_dev *pdev, > enum xe_engine_class class, > const u32 **cs); > @@ -44,6 +45,7 @@ static inline bool xe_configfs_media_gt_allowed(struct pci_dev *pdev) { return t > static inline u64 xe_configfs_get_engines_allowed(struct pci_dev *pdev) { return U64_MAX; } > static inline bool xe_configfs_get_psmi_enabled(struct pci_dev *pdev) { return false; } > static inline bool xe_configfs_get_enable_multi_queue(struct pci_dev *pdev) { return true; } > +static inline bool xe_configfs_get_bad_page_reservation(struct pci_dev *pdev) { return true; } > static inline u32 xe_configfs_get_ctx_restore_mid_bb(struct pci_dev *pdev, > enum xe_engine_class class, > const u32 **cs) { return 0; } > diff --git a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c > index 370bcf50c7f7..6280886e2ebb 100644 > --- a/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c > +++ b/drivers/gpu/drm/xe/xe_ttm_vram_mgr.c > @@ -13,6 +13,7 @@ > > #include "regs/xe_regs.h" > #include "xe_bo.h" > +#include "xe_configfs.h" > #include "xe_device.h" > #include "xe_exec_queue.h" > #include "xe_lrc.h" > @@ -792,6 +793,7 @@ int xe_ttm_vram_handle_addr_fault(struct xe_device *xe, u64 addr) > struct xe_ttm_vram_mgr *vram_mgr; > struct xe_vram_region *vr; > struct gpu_buddy *mm; > + bool policy; > > vr = xe_ttm_vram_addr_to_region(xe, addr); > if (IS_ERR(vr)) { > @@ -811,6 +813,14 @@ int xe_ttm_vram_handle_addr_fault(struct xe_device *xe, u64 addr) > vram_mgr = &vr->ttm; > mm = &vram_mgr->mm; > > + policy = xe_configfs_get_bad_page_reservation(to_pci_dev(xe->drm.dev)); > + if (!policy) { > + drm_err(&xe->drm, "0x%llx is reported as corrupted address by HW\n", > + addr); > + /* Let RAS report to FW to drop addr from SRAM queue */ > + return -EOPNOTSUPP; > + } > + > /* Reserve page at address */ > return xe_ttm_vram_reserve_page_at_addr(xe, addr, vram_mgr, mm); > }