[PATCH v13 03/12] famfs: Add daxdev table and dax notify_failure support

John Groves <[email protected]>
Newsgroups dev.linux.lists.nvdimm,dev.linux.lists.fuse-devel,org.kernel.vger.linux-cxl,org.kernel.vger.linux-doc,org.kernel.vger.linux-fsdevel,org.kernel.vger.linux-kernel
Message-ID <0100019fed5945de-89b374bb-2792-4bbe-acc5-a806f6e1b309-000000@email.amazonses.com>
From: John Groves <[email protected]>

Famfs file systems can span multiple dax devices, and daxdevs are stored
in the daxdev_table. This adds the basic table structure, primtives and
serialization code. Famfs file extents reference daxdevs by index, which
is a cluster invariant maintained by user space.

We also add dax_holder_operations and a notify_failure handler, which
is necessary to properly "open" a famfs-mode daxdev.

Signed-off-by: John Groves <[email protected]>
---
 fs/famfs/famfs_inode.c    | 240 ++++++++++++++++++++++++++++++++++++++
 fs/famfs/famfs_internal.h |  46 ++++++++
 2 files changed, 286 insertions(+)

diff --git a/fs/famfs/famfs_inode.c b/fs/famfs/famfs_inode.c
index 5a13903da61b..5735d8d1900b 100644
--- a/fs/famfs/famfs_inode.c
+++ b/fs/famfs/famfs_inode.c
@@ -76,6 +76,231 @@ famfs_get_inode(
 /*
  * famfs dax_operations (for famfs-mode dax)
  */
+static void famfs_set_daxdev_err(struct famfs_fs_info *fsi,
+				 struct dax_device *dax_devp);
+
+static int
+famfs_dax_notify_failure(
+	struct dax_device *dax_dev,
+	u64 offset,
+	u64 len,
+	int mf_flags)
+{
+	struct super_block *sb = dax_holder(dax_dev);
+	struct famfs_fs_info *fsi = sb->s_fs_info;
+
+	pr_err("%s: offset=%lld len=%llu flags=%x\n", __func__,
+	       offset, len, mf_flags);
+
+	/*
+	 * Record the error on the specific daxdev and, near-term, shut the
+	 * mount down: famfs_set_daxdev_err() also sets fsi->deverror so
+	 * subsequent famfs operations fail. The resolver's per-daxdev
+	 * famfs_dax_err() check remains and can make this finer later.
+	 */
+	famfs_set_daxdev_err(fsi, dax_dev);
+
+	return 0;
+}
+
+static const struct dax_holder_operations famfs_dax_holder_ops = {
+	.notify_failure		= famfs_dax_notify_failure,
+};
+
+/*
+ * Allocate the daxdev table on first use (idempotent via cmpxchg).
+ */
+int
+famfs_devlist_alloc(struct famfs_fs_info *fsi)
+{
+	struct famfs_dax_devlist *devlist;
+
+	if (fsi->dax_devlist)
+		return 0;
+
+	devlist = kcalloc(1, sizeof(*devlist), GFP_KERNEL);
+	if (!devlist)
+		return -ENOMEM;
+
+	devlist->nslots = FAMFS_MAX_DAXDEVS;
+	devlist->devlist = kcalloc(FAMFS_MAX_DAXDEVS, sizeof(struct famfs_daxdev),
+				   GFP_KERNEL);
+	if (!devlist->devlist) {
+		kfree(devlist);
+		return -ENOMEM;
+	}
+
+	/* If another thread allocated it first, drop ours */
+	if (cmpxchg(&fsi->dax_devlist, NULL, devlist) != NULL) {
+		kfree(devlist->devlist);
+		kfree(devlist);
+	}
+
+	return 0;
+}
+
+/*
+ * famfs_install_daxdev() - exclusively acquire a resolved daxdev and publish
+ * it in the table at @index. Slot 0 is the mount primary; slots 1..n come from
+ * the daxdev-open ioctl.
+ *
+ * Serializes with concurrent installers under devlist_sem and rechecks
+ * ->valid, so re-registering an already-installed slot is idempotent. A daxdev
+ * is entered in the table only once it has been exclusively acquired via
+ * fs_dax_get() (with the super_block as the holder); on failure the
+ * dax_dev_find() reference is released and the slot is left invalid. @name may
+ * be NULL (the ioctl path passes no pathname).
+ */
+int
+famfs_install_daxdev(
+	struct famfs_fs_info *fsi,
+	struct super_block *sb,
+	u64 index,
+	dev_t devno,
+	const char *name)
+{
+	struct famfs_daxdev *daxdev;
+	int rc = 0;
+
+	if (index >= fsi->dax_devlist->nslots) {
+		pr_debug("%s: index(%llu) >= nslots(%d)\n",
+		       __func__, index, fsi->dax_devlist->nslots);
+		return -EINVAL;
+	}
+
+	scoped_guard(rwsem_write, &fsi->devlist_sem) {
+		daxdev = &fsi->dax_devlist->devlist[index];
+
+		/* Installed already by a concurrent (or repeated) open */
+		if (daxdev->valid)
+			return 0;
+
+		/*
+		 * A prior attempt already determined this daxdev cannot be
+		 * exclusively acquired (see the fs_dax_get() failure handling
+		 * below). Don't thrash on fs_dax_get(); fail fast.
+		 */
+		if (daxdev->dax_err)
+			return -EIO;
+
+		daxdev->devp = dax_dev_find(devno);
+		if (!daxdev->devp) {
+			pr_debug("%s: device %u:%u not found or not dax\n",
+				__func__, MAJOR(devno), MINOR(devno));
+			return -ENODEV;
+		}
+
+		rc = fs_dax_get(daxdev->devp, sb, &famfs_dax_holder_ops);
+		if (rc) {
+			/*
+			 * Distinguish a lost race from a real failure. -EBUSY
+			 * with the daxdev already held by *this* super_block
+			 * means a concurrent acquire won and will publish the
+			 * slot valid: not an error, and must not be cached as
+			 * dax_err. Any other failure is permanent for this
+			 * mount, so record dax_err to stop re-acquiring it.
+			 */
+			if (!(rc == -EBUSY && dax_holder(daxdev->devp) == sb)) {
+				pr_debug("%s: fs_dax_get(%u:%u) failed rc=%d\n",
+				       __func__, MAJOR(devno), MINOR(devno), rc);
+				daxdev->dax_err = true;
+			}
+			put_dax(daxdev->devp);
+			daxdev->devp = NULL;
+			return rc;
+		}
+
+		daxdev->devno = devno;
+		if (name) {
+			daxdev->name = kstrdup(name, GFP_KERNEL);
+			if (!daxdev->name) {
+				fs_put_dax(daxdev->devp, sb);
+				put_dax(daxdev->devp);
+				daxdev->devp = NULL;
+				return -ENOMEM;
+			}
+		}
+
+		wmb(); /* All other fields must be visible before valid */
+		daxdev->valid = 1;
+	}
+
+	return 0;
+}
+
+/*
+ * Release every daxdev in the table and free it. Detach the table under
+ * devlist_sem so a notify_failure racing teardown either runs first against
+ * the live table or observes dax_devlist == NULL and bails.
+ */
+static
+void famfs_devlist_free(
+	struct famfs_fs_info *fsi,
+	struct super_block *sb)
+{
+	struct famfs_dax_devlist *devlist __free(kfree) = NULL;
+	int i;
+
+	scoped_guard(rwsem_write, &fsi->devlist_sem) {
+		devlist = fsi->dax_devlist;
+		fsi->dax_devlist = NULL;
+	}
+
+	if (!devlist || !devlist->devlist)
+		return;
+
+	for (i = 0; i < devlist->nslots; i++) {
+		struct famfs_daxdev *dd = &devlist->devlist[i];
+
+		if (!dd->valid)
+			continue;
+
+		if (dd->devp) {
+			if (!dd->dax_err)
+				fs_put_dax(dd->devp, sb);
+			put_dax(dd->devp);
+		}
+		kfree(dd->name);
+	}
+	kfree(devlist->devlist);
+}
+
+/*
+ * Record a memory error on the daxdev matching @dax_devp. Searches the table
+ * under the write lock (which serializes against famfs_devlist_free()).
+ */
+static void
+famfs_set_daxdev_err(
+	struct famfs_fs_info *fsi,
+	struct dax_device *dax_devp)
+{
+	int i;
+
+	scoped_guard(rwsem_write, &fsi->devlist_sem) {
+		if (!fsi->dax_devlist)
+			return;
+		for (i = 0; i < fsi->dax_devlist->nslots; i++) {
+			struct famfs_daxdev *dd = &fsi->dax_devlist->devlist[i];
+
+			if (!dd->valid || dd->devp != dax_devp)
+				continue;
+
+			dd->error = true;
+			/*
+			 * Near-term policy: any daxdev memory error shuts down
+			 * the whole mount. Finer per-daxdev handling (via
+			 * famfs_dax_err() in the resolver) already exists and
+			 * can supersede this later.
+			 */
+			fsi->deverror = true;
+			pr_err("%s: memory error on daxdev %s (%d)\n",
+			       __func__, dd->name, i);
+			return;
+		}
+	}
+	pr_debug("%s: memory error on unrecognized daxdev\n", __func__);
+}
+
 /*****************************************************************************
  * fs_context_operations
  */
@@ -159,6 +384,18 @@ famfs_get_tree(struct fs_context *fc)
 		famfs_fill_super(sb, fc);
 	}
 
+	/* Install the primary daxdev (from the mount device) at slot 0 */
+	err = famfs_devlist_alloc(fsi);
+	if (err)
+		goto deactivate_out;
+
+	err = famfs_install_daxdev(fsi, sb, 0, daxdevno, fc->source);
+	if (err) {
+		pr_err("%s: failed to install primary daxdev %s\n",
+		       __func__, fc->source);
+		goto deactivate_out;
+	}
+
 	inode = famfs_get_inode(sb, NULL, S_IFDIR | fsi->mount_opts.mode, 0);
 	sb->s_root = d_make_root(inode);
 	if (!sb->s_root) {
@@ -247,6 +484,7 @@ famfs_init_fs_context(struct fs_context *fc)
 	if (!fsi)
 		return -ENOMEM;
 
+	init_rwsem(&fsi->devlist_sem);
 	fsi->mount_opts.mode = FAMFS_DEFAULT_MODE;
 	fc->s_fs_info        = fsi;
 	fc->ops              = &famfs_context_ops;
@@ -258,6 +496,8 @@ famfs_kill_sb(struct super_block *sb)
 {
 	struct famfs_fs_info *fsi = sb->s_fs_info;
 
+	famfs_devlist_free(fsi, sb);
+
 	kill_char_super(sb);
 
 	kfree(fsi);
diff --git a/fs/famfs/famfs_internal.h b/fs/famfs/famfs_internal.h
index 378544b1aabe..37f667b2b79a 100644
--- a/fs/famfs/famfs_internal.h
+++ b/fs/famfs/famfs_internal.h
@@ -11,22 +11,68 @@
 #ifndef FAMFS_INTERNAL_H
 #define FAMFS_INTERNAL_H
 
+#include <linux/rwsem.h>
+#include <linux/bits.h>
+#include <linux/build_bug.h>
+
 struct famfs_mount_opts {
 	umode_t mode;
 };
 
+/*
+ * famfs_daxdev - one entry in the per-superblock daxdev table
+ *
+ * @valid:   slot is populated and the daxdev has been exclusively acquired
+ * @error:   dax reported a memory error (probably poison) via notify_failure
+ * @dax_err: fs_dax_get() failed for this daxdev
+ * @devno:   dax device dev_t
+ * @devp:    the acquired dax_device
+ * @name:    dax device path (may be NULL for ioctl-registered daxdevs)
+ */
+struct famfs_daxdev {
+	bool valid;
+	bool error;
+	bool dax_err;
+	dev_t devno;
+	struct dax_device *devp;
+	char *name;
+};
+
+/*
+ * The daxdev index space (and thus this table) is capped at 64 so the set of
+ * daxdev indices referenced by a file's fmap fits in a u64 bitmap.
+ */
+#define FAMFS_MAX_DAXDEVS 64
+static_assert(BITS_PER_TYPE(u64) >= FAMFS_MAX_DAXDEVS);
+
+/*
+ * famfs_dax_devlist - the per-superblock table of famfs_daxdev's. Slot 0 is
+ * the primary daxdev supplied at mount; slots 1..n are registered via ioctl.
+ */
+struct famfs_dax_devlist {
+	int nslots;
+	struct famfs_daxdev *devlist;
+};
+
 /**
  * @famfs_fs_info
  *
  * @mount_opts:  The mount options
  * @deverror:    True if the dax device has called our notify_failure entry
  *               point, or if other "shutdown" conditions exist
+ * @dax_devlist: Table of backing daxdevs (slot 0 is the mount primary)
+ * @devlist_sem: Serializes installs into, and teardown of, @dax_devlist
  */
 struct famfs_fs_info {
 	struct famfs_mount_opts   mount_opts;
 	bool                      deverror;
+	struct famfs_dax_devlist *dax_devlist;
+	struct rw_semaphore       devlist_sem;
 };
 
 int famfs_lookup_daxdev(const char *pathname, dev_t *devno);
+int famfs_devlist_alloc(struct famfs_fs_info *fsi);
+int famfs_install_daxdev(struct famfs_fs_info *fsi, struct super_block *sb,
+			 u64 index, dev_t devno, const char *name);
 
 #endif /* FAMFS_INTERNAL_H */
-- 
2.53.0
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.