[PATCH v2 12/23] NFSv4/flexfiles: Implement in-place device re-resolve on CHANGE

Benjamin Coddington <[email protected]>
Newsgroups org.kernel.vger.linux-nfs
Message-ID <631adecc29c60d90a0063c3dee342385d50dad0a.1787327939.git.bcodding@hammerspace.com>
Implement the reresolve_deviceid hook: walk the layout's mirrors and,
for every stripe whose raw deviceid matches the changed one, exchange
the pinned device node out for NULL.  In-flight I/O completes on the old
node through its own reference; the next I/O to the stripe re-resolves
via ff_layout_get_mirror_ds() and picks up the server's new mapping
with a fresh GETDEVICEINFO.  The un-pinned references are handed back
on the walker's list to be put outside the locks.

Matching uses the raw deviceid decoded from the layout (dss[].devid), so
a stripe that was never resolved, or already exchanged, is left alone.
A stripe whose resolution previously failed holds an ERR_PTR sentinel
rather than a node; that is cleared too, so the stripe retries
GETDEVICEINFO on its next I/O.  Nothing dispatches CHANGE to the walker
yet: no behavior change.

Assisted-by: Claude:claude-fable-5
Signed-off-by: Benjamin Coddington <[email protected]>
---
 fs/nfs/flexfilelayout/flexfilelayout.c | 40 ++++++++++++++++++++++++++
 1 file changed, 40 insertions(+)

diff --git a/fs/nfs/flexfilelayout/flexfilelayout.c b/fs/nfs/flexfilelayout/flexfilelayout.c
index 3572324630df..46ca58e8e96b 100644
--- a/fs/nfs/flexfilelayout/flexfilelayout.c
+++ b/fs/nfs/flexfilelayout/flexfilelayout.c
@@ -2508,6 +2508,45 @@ static void ff_layout_cancel_io(struct pnfs_layout_segment *lseg)
 	}
 }
 
+/*
+ * Un-pin every stripe node resolved from @id: in-flight I/O drains on the
+ * old node through its own reference, the next I/O re-resolves.
+ */
+static void ff_layout_reresolve_deviceid(struct pnfs_layout_hdr *lo,
+					 const struct nfs4_deviceid *id,
+					 bool immediate,
+					 struct list_head *head)
+{
+	struct nfs4_flexfile_layout *flo = FF_LAYOUT_FROM_HDR(lo);
+	struct nfs4_ff_layout_mirror *mirror;
+	struct nfs4_ff_layout_ds *old;
+	struct nfs4_deviceid_put *put;
+	u32 dss_id;
+
+	list_for_each_entry(mirror, &flo->mirrors, mirrors) {
+		for (dss_id = 0; dss_id < mirror->dss_count; dss_id++) {
+			if (memcmp(&mirror->dss[dss_id].devid, id,
+				   sizeof(*id)) != 0)
+				continue;
+			/* Allocate before un-pinning: on failure the reference
+			 * stays put rather than being dropped here, where the
+			 * final put may not sleep.
+			 */
+			put = kzalloc_obj(*put, GFP_ATOMIC);
+			if (!put)
+				continue;
+			old = unrcu_pointer(
+				xchg(&mirror->dss[dss_id].mirror_ds, NULL));
+			if (IS_ERR_OR_NULL(old)) {
+				kfree(put);
+				continue;
+			}
+			put->dev = &old->id_node;
+			list_add(&put->node, head);
+		}
+	}
+}
+
 static struct pnfs_ds_commit_info *
 ff_layout_get_ds_info(struct inode *inode)
 {
@@ -3089,6 +3128,7 @@ static struct pnfs_layoutdriver_type flexfilelayout_type = {
 	.pg_write_ops		= &ff_layout_pg_write_ops,
 	.get_ds_info		= ff_layout_get_ds_info,
 	.free_deviceid_node	= ff_layout_free_deviceid_node,
+	.reresolve_deviceid	= ff_layout_reresolve_deviceid,
 	.read_pagelist		= ff_layout_read_pagelist,
 	.write_pagelist		= ff_layout_write_pagelist,
 	.alloc_deviceid_node    = ff_layout_alloc_deviceid_node,
-- 
2.53.0
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.