[PATCH V11 5/9] famfs_fuse: register fs-dax daxdevs via FUSE_DEV_IOC_DAXDEV_OPEN

John Groves <[email protected]>
Newsgroups org.kernel.vger.linux-cxl,dev.linux.lists.fuse-devel,dev.linux.lists.nvdimm,org.kernel.vger.linux-doc,org.kernel.vger.linux-fsdevel,org.kernel.vger.linux-kernel
Message-ID <0100019f7da1a534-2fc6658c-4f30-4cc6-b010-5e3f8fd7164b-000000@email.amazonses.com>
From: John Groves <[email protected]>

Add a dedicated ioctl, FUSE_DEV_IOC_DAXDEV_OPEN, by which a fuse server
registers the devdax devices that back an fs-dax (famfs) filesystem: the
server hands the kernel an fd to a /dev/daxN.N device plus its
cluster-invariant famfs index.

A dedicated ioctl, rather than overloading FUSE_DEV_IOC_BACKING_OPEN,
avoids a dependency on CONFIG_FUSE_PASSTHROUGH and the backing-file
machinery, which famfs does not use.

- struct fuse_backing_map is reused as the argument; the index rides on
  the reserved 'padding' field (daxdev_index).
- fuse_dev_ioctl_daxdev_open() copies the map and calls
  famfs_daxdev_open(), which resolves the fd to a dax device by its inode
  i_rdev. The daxdev table store is added in the following patch.

Access control:

- The ioctl is gated on famfs mode (enabled at FUSE_INIT only when the
  server held CAP_SYS_RAWIO), and additionally re-checks
  capable(CAP_SYS_RAWIO) on each call. The famfs-mode flag attests only
  to the session founder's privilege; the fuse device fd may be inherited
  across fork() or SCM_RIGHTS-passed to a less-privileged task, so the
  capability is re-checked against the task actually performing the
  registration.
- famfs_daxdev_open() resolves the fd with fget(), not fget_raw(), so
  O_PATH fds are rejected. A successful lookup then proves the caller
  holds an fd from a real open() of the daxdev -- i.e. it already passed
  may_open_dev() and the device node's DAC checks -- rather than an
  O_PATH reference that bypasses them.

Signed-off-by: John Groves <[email protected]>
---
 fs/fuse/dev.c             |  31 ++++
 fs/fuse/dev.h             |   1 +
 fs/fuse/famfs.c           | 330 ++++++++++++++++++++++++++++++++++++++
 fs/fuse/famfs_kfmap.h     |  27 ++++
 fs/fuse/fuse_i.h          |  28 ++++
 fs/fuse/inode.c           |   7 +-
 include/uapi/linux/fuse.h |   7 +-
 7 files changed, 429 insertions(+), 2 deletions(-)

diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c
index 5763a7cd3b37..3e6aa15e0346 100644
--- a/fs/fuse/dev.c
+++ b/fs/fuse/dev.c
@@ -2330,6 +2330,34 @@ static long fuse_dev_ioctl_backing_close(struct file *file, __u32 __user *argp)
 	return fuse_backing_close(fud->chan->conn, backing_id);
 }
 
+static long fuse_dev_ioctl_daxdev_open(struct file *file,
+				       struct fuse_backing_map __user *argp)
+{
+	struct fuse_dev *fud = fuse_get_dev(file);
+	struct fuse_backing_map map;
+
+	if (IS_ERR(fud))
+		return PTR_ERR(fud);
+
+	if (!IS_ENABLED(CONFIG_FUSE_FAMFS_DAX))
+		return -EOPNOTSUPP;
+
+	/*
+	 * The famfs-mode gate (fc->famfs_iomap) lives in famfs_daxdev_open(),
+	 * which has the full fuse_conn definition. Here, re-check CAP_SYS_RAWIO
+	 * against the task performing the registration: famfs mode being enabled
+	 * only attests that the session founder held it at FUSE_INIT, and the
+	 * fuse device fd may have been passed to a less-privileged process.
+	 */
+	if (!capable(CAP_SYS_RAWIO))
+		return -EPERM;
+
+	if (copy_from_user(&map, argp, sizeof(map)))
+		return -EFAULT;
+
+	return famfs_daxdev_open(fud->chan->conn, &map);
+}
+
 static long fuse_dev_ioctl_sync_init(struct file *file)
 {
 	struct fuse_dev *fud = fuse_file_to_fud(file);
@@ -2359,6 +2387,9 @@ static long fuse_dev_ioctl(struct file *file, unsigned int cmd,
 	case FUSE_DEV_IOC_SYNC_INIT:
 		return fuse_dev_ioctl_sync_init(file);
 
+	case FUSE_DEV_IOC_DAXDEV_OPEN:
+		return fuse_dev_ioctl_daxdev_open(file, argp);
+
 	default:
 		return -ENOTTY;
 	}
diff --git a/fs/fuse/dev.h b/fs/fuse/dev.h
index aed69fd14c41..545940b635cc 100644
--- a/fs/fuse/dev.h
+++ b/fs/fuse/dev.h
@@ -87,6 +87,7 @@ int fuse_notify(struct fuse_conn *fc, enum fuse_notify_code code,
 
 int fuse_backing_open(struct fuse_conn *fc, struct fuse_backing_map *map);
 int fuse_backing_close(struct fuse_conn *fc, int backing_id);
+int famfs_daxdev_open(struct fuse_conn *fc, struct fuse_backing_map *map);
 
 int fuse_copy_one(struct fuse_copy_state *cs, void *val, unsigned size);
 int fuse_copy_folio(struct fuse_copy_state *cs, struct folio **foliop,
diff --git a/fs/fuse/famfs.c b/fs/fuse/famfs.c
index 8f7ee7d6151b..a2a7dd631dc0 100644
--- a/fs/fuse/famfs.c
+++ b/fs/fuse/famfs.c
@@ -11,6 +11,7 @@
 
 #include <linux/cleanup.h>
 #include <linux/fs.h>
+#include <linux/file.h>
 #include <linux/mm.h>
 #include <linux/dax.h>
 #include <linux/iomap.h>
@@ -22,6 +23,331 @@
 #include "famfs_kfmap.h"
 #include "fuse_i.h"
 
+static void famfs_set_daxdev_err(
+	struct fuse_conn *fc, struct dax_device *dax_devp);
+
+static int
+famfs_dax_notify_failure(struct dax_device *dax_devp, u64 offset,
+			u64 len, int mf_flags)
+{
+	struct fuse_conn *fc = dax_holder(dax_devp);
+
+	famfs_set_daxdev_err(fc, dax_devp);
+
+	return 0;
+}
+
+static const struct dax_holder_operations famfs_fuse_dax_holder_ops = {
+	.notify_failure		= famfs_dax_notify_failure,
+};
+
+/*****************************************************************************/
+
+/*
+ * famfs_teardown()
+ *
+ * Deallocate famfs metadata for a fuse_conn
+ */
+void
+famfs_teardown(struct fuse_conn *fc)
+{
+	struct famfs_dax_devlist *devlist __free(kfree) = NULL;
+	int i;
+
+	/*
+	 * Detach the table under the same lock famfs_set_daxdev_err() takes, so
+	 * a notify_failure racing teardown either runs first against the live
+	 * table or observes dax_devlist == NULL and bails, rather than
+	 * dereferencing it after we clear it. The daxdev holders are dropped
+	 * below, after which no further notify_failure can arrive.
+	 */
+	scoped_guard(rwsem_write, &fc->famfs_devlist_sem) {
+		devlist = fc->dax_devlist;
+		fc->dax_devlist = NULL;
+	}
+
+	if (!devlist)
+		return;
+
+	if (!devlist->devlist)
+		return;
+
+	/* Close & release all the daxdevs in our table */
+	for (i = 0; i < devlist->nslots; i++) {
+		struct famfs_daxdev *dd = &devlist->devlist[i];
+
+		if (!dd->valid)
+			continue;
+
+		/* Only call fs_put_dax if fs_dax_get succeeded */
+		if (dd->devp) {
+			if (!dd->dax_err)
+				fs_put_dax(dd->devp, fc);
+			put_dax(dd->devp);
+		}
+
+		kfree(dd->name);
+	}
+	kfree(devlist->devlist);
+}
+
+/* Allocate the daxdev table on first use (idempotent via cmpxchg) */
+static int famfs_devlist_alloc(struct fuse_conn *fc)
+{
+	struct famfs_dax_devlist *devlist;
+
+	if (fc->dax_devlist)
+		return 0;
+
+	devlist = kcalloc(1, sizeof(*devlist), GFP_KERNEL);
+	if (!devlist)
+		return -ENOMEM;
+
+	devlist->nslots = MAX_DAXDEVS;
+	devlist->devlist = kcalloc(MAX_DAXDEVS, sizeof(struct famfs_daxdev),
+				   GFP_KERNEL);
+	if (!devlist->devlist) {
+		kfree(devlist);
+		return -ENOMEM;
+	}
+
+	/* If another thread allocated it first, drop ours */
+	if (cmpxchg(&fc->dax_devlist, NULL, devlist) != NULL) {
+		kfree(devlist->devlist);
+		kfree(devlist);
+	}
+
+	return 0;
+}
+
+/*
+ * famfs_install_daxdev() - exclusively acquire a resolved daxdev and publish
+ * it in the table at @index. Shared by the GET_DAXDEV (pull) and
+ * DAXDEV_OPEN (push) registration paths.
+ *
+ * Serializes with concurrent installers under famfs_devlist_sem and rechecks
+ * ->valid. A daxdev is entered in the table only once it has been exclusively
+ * acquired via fs_dax_get(); on failure the dax_dev_get() reference is
+ * released and the slot is left invalid, so the referencing fmap is rejected
+ * rather than mapped without an exclusive holder. @name may be NULL (the push
+ * path passes no pathname).
+ */
+static int famfs_install_daxdev(struct fuse_conn *fc, u64 index, dev_t devno,
+				const char *name)
+{
+	struct famfs_daxdev *daxdev;
+	int rc = 0;
+
+	if (index >= fc->dax_devlist->nslots) {
+		pr_err("%s: index(%llu) >= nslots(%d)\n",
+		       __func__, index, fc->dax_devlist->nslots);
+		return -EINVAL;
+	}
+
+	scoped_guard(rwsem_write, &fc->famfs_devlist_sem) {
+		daxdev = &fc->dax_devlist->devlist[index];
+
+		/* Installed already by a concurrent push/pull */
+		if (daxdev->valid)
+			return 0;
+
+		/*
+		 * A prior attempt already determined this daxdev cannot be
+		 * exclusively acquired (see the fs_dax_get() failure handling
+		 * below). Don't thrash on GET_DAXDEV/fs_dax_get(); fail fast.
+		 */
+		if (daxdev->dax_err)
+			return -EIO;
+
+		/*
+		 * Temporary: dax_dev_get() is the exported upstream lookup, but
+		 * unlike dax_dev_find() it allocates for any dev_t and does not
+		 * reject non-dax devices. Restore dax_dev_find() (and that
+		 * rejection) once it is upstream.
+		 */
+		daxdev->devp = dax_dev_get(devno);
+		if (!daxdev->devp) {
+			pr_warn("%s: device %u:%u not found or not dax\n",
+				__func__, MAJOR(devno), MINOR(devno));
+			return -ENODEV;
+		}
+
+		rc = fs_dax_get(daxdev->devp, fc, &famfs_fuse_dax_holder_ops);
+		if (rc) {
+			/*
+			 * Distinguish a lost race from a real failure. -EBUSY
+			 * with the daxdev already held by *this* fuse_conn
+			 * means a concurrent acquire won and will publish the
+			 * slot valid: not an error, and must not be cached as
+			 * dax_err. Any other failure (foreign holder, not a dax
+			 * device, wrong driver type) is permanent for this
+			 * connection, so record dax_err to stop re-fetching and
+			 * re-acquiring it.
+			 */
+			if (!(rc == -EBUSY && dax_holder(daxdev->devp) == fc)) {
+				pr_err("%s: fs_dax_get(%u:%u) failed rc=%d\n",
+				       __func__, MAJOR(devno), MINOR(devno), rc);
+				daxdev->dax_err = true;
+			}
+			put_dax(daxdev->devp);
+			daxdev->devp = NULL;
+			return rc;
+		}
+
+		daxdev->devno = devno;
+		if (name) {
+			daxdev->name = kstrdup(name, GFP_KERNEL);
+			if (!daxdev->name) {
+				fs_put_dax(daxdev->devp, fc);
+				put_dax(daxdev->devp);
+				daxdev->devp = NULL;
+				return -ENOMEM;
+			}
+		}
+
+		wmb(); /* All other fields must be visible before valid */
+		daxdev->valid = 1;
+	}
+
+	return 0;
+}
+
+/**
+ * famfs_daxdev_open() - Register a daxdev via FUSE_DEV_IOC_DAXDEV_OPEN
+ * @fc:   fuse_conn
+ * @map:  fuse_backing_map; @map->fd is an fd to the devdax device and
+ *        @map->daxdev_index is the (cluster-invariant) famfs index.
+ *
+ * The server pushes a daxdev to the kernel by reference (an fd), rather than
+ * the kernel pulling it by name via GET_DAXDEV. The resolved daxdev is
+ * exclusively acquired and entered in the table at @map->daxdev_index.
+ *
+ * Return: 0=success
+ *         -errno=failure
+ */
+int famfs_daxdev_open(struct fuse_conn *fc, struct fuse_backing_map *map)
+{
+	struct inode *inode;
+	struct file *file;
+	dev_t devno;
+	int rc;
+
+	/* Only fs-dax (famfs) mode accepts daxdev registration */
+	if (!fc->famfs_iomap)
+		return -EOPNOTSUPP;
+
+	file = fget(map->fd);
+	if (!file)
+		return -EBADF;
+
+	inode = file_inode(file);
+	if (!S_ISCHR(inode->i_mode)) {
+		fput(file);
+		return -EINVAL;
+	}
+	devno = inode->i_rdev;
+	fput(file);
+
+	rc = famfs_devlist_alloc(fc);
+	if (rc)
+		return rc;
+
+	rc = famfs_install_daxdev(fc, map->daxdev_index, devno, NULL);
+	if (rc)
+		pr_err("%s: failed to install daxdev\n", __func__);
+
+	return rc;
+}
+
+/**
+ * famfs_check_daxdev_table() - Verify an fmap's referenced daxdevs are installed
+ * @fm:   fuse_mount
+ * @meta: famfs_file_meta, in-memory format, built from a GET_FMAP response
+ *
+ * Called for each new file fmap. Every daxdev the fmap references must already
+ * be installed in the table, having been pushed in via FUSE_DEV_IOC_DAXDEV_OPEN
+ * before any file that uses it is accessed. If any referenced daxdev is not
+ * present, the fmap is rejected so the file is never mapped against a daxdev
+ * that has no exclusive holder.
+ *
+ * Return: 0=success (all referenced daxdevs present)
+ *         <0=a referenced daxdev is missing from the table
+ */
+static int
+famfs_check_daxdev_table(
+	struct fuse_mount *fm,
+	const struct famfs_file_meta *meta)
+{
+	struct fuse_conn *fc = fm->fc;
+	int nmissing = 0;
+	int err;
+
+	err = famfs_devlist_alloc(fc);
+	if (err)
+		return err;
+
+	/* Count missing daxdevs while holding the reader lock */
+	scoped_guard(rwsem_read, &fc->famfs_devlist_sem) {
+		unsigned long i;
+
+		for_each_set_bit(i, (unsigned long *)&meta->dev_bitmap,
+				 MAX_DAXDEVS) {
+			struct famfs_daxdev *dd = &fc->dax_devlist->devlist[i];
+
+			/*
+			 * Skip daxdevs already installed (valid) or already
+			 * known to be unusable (dax_err). Re-fetching either
+			 * just thrashes on GET_DAXDEV and fs_dax_get().
+			 */
+			if (!dd->valid && !dd->dax_err)
+				nmissing++;
+		}
+	}
+
+	if (nmissing > 0) {
+		/* this file referenced at least one daxdev that is not in
+		 * the table. Daxdevs must be known before any file that
+		 * uses them is accessed
+		 */
+		pr_err("%s: %d missing daxdev(s)\n", __func__, nmissing);
+		return -ENODEV;
+	}
+
+	return 0;
+}
+
+static void
+famfs_set_daxdev_err(
+	struct fuse_conn *fc,
+	struct dax_device *dax_devp)
+{
+	int i;
+
+	/*
+	 * Search the list by dax_devp under the write lock: we set dd->error,
+	 * and it serializes against famfs_teardown() clearing the table.
+	 */
+	scoped_guard(rwsem_write, &fc->famfs_devlist_sem) {
+		if (!fc->dax_devlist)
+			return;
+		for (i = 0; i < fc->dax_devlist->nslots; i++) {
+			if (fc->dax_devlist->devlist[i].valid) {
+				struct famfs_daxdev *dd;
+
+				dd = &fc->dax_devlist->devlist[i];
+				if (dd->devp != dax_devp)
+					continue;
+
+				dd->error = true;
+
+				pr_err("%s: memory error on daxdev %s (%d)\n",
+				       __func__, dd->name, i);
+				return;
+			}
+		}
+	}
+	pr_err("%s: memory err on unrecognized daxdev\n", __func__);
+}
 
 /***************************************************************************/
 
@@ -228,6 +554,10 @@ famfs_file_init_dax(
 	if (rc)
 		goto errout;
 
+	/* Make sure this fmap doesn't reference any unknown daxdevs */
+	if (famfs_check_daxdev_table(fm, meta))
+		meta->error = true;
+
 	/* Publish the famfs metadata on fi->famfs_meta */
 	inode_lock(inode);
 
diff --git a/fs/fuse/famfs_kfmap.h b/fs/fuse/famfs_kfmap.h
index d87b065e8ac8..6b78e78325c8 100644
--- a/fs/fuse/famfs_kfmap.h
+++ b/fs/fuse/famfs_kfmap.h
@@ -61,4 +61,31 @@ struct famfs_file_meta {
 	struct famfs_meta_simple_ext  *se;
 };
 
+/*
+ * famfs_daxdev - tracking struct for a daxdev within a famfs file system
+ *
+ * This is the in-memory daxdev metadata that is populated by parsing
+ * the responses to GET_FMAP messages
+ */
+struct famfs_daxdev {
+	/* Include dev uuid? */
+	bool valid;
+	bool error; /* Dax has reported a memory error (probably poison) */
+	bool dax_err; /* fs_dax_get() failed */
+	dev_t devno;
+	struct dax_device *devp;
+	char *name;
+};
+
+#define MAX_DAXDEVS 24
+
+/*
+ * famfs_dax_devlist - list of famfs_daxdev's
+ */
+struct famfs_dax_devlist {
+	int nslots;
+	int ndevs;
+	struct famfs_daxdev *devlist;
+};
+
 #endif /* FAMFS_KFMAP_H */
diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h
index 46a7040b38dc..5394aae9dbac 100644
--- a/fs/fuse/fuse_i.h
+++ b/fs/fuse/fuse_i.h
@@ -773,6 +773,11 @@ struct fuse_conn {
 	/** @backing_files_map: IDR for backing files ids */
 	struct idr backing_files_map;
 #endif
+
+#if IS_ENABLED(CONFIG_FUSE_FAMFS_DAX)
+	struct rw_semaphore famfs_devlist_sem;
+	struct famfs_dax_devlist *dax_devlist;
+#endif
 };
 
 /*
@@ -1352,6 +1357,8 @@ int famfs_file_init_dax(struct fuse_mount *fm,
 			size_t fmap_size);
 void __famfs_meta_free(void *map);
 
+void famfs_teardown(struct fuse_conn *fc);
+
 /* Set fi->famfs_meta = NULL regardless of prior value */
 static inline void famfs_meta_init(struct fuse_inode *fi)
 {
@@ -1373,6 +1380,11 @@ static inline void famfs_meta_free(struct fuse_inode *fi)
 	}
 }
 
+static inline void famfs_init_devlist_sem(struct fuse_conn *fc)
+{
+	init_rwsem(&fc->famfs_devlist_sem);
+}
+
 static inline int fuse_file_famfs(struct fuse_inode *fi)
 {
 	return (READ_ONCE(fi->famfs_meta) != NULL);
@@ -1380,8 +1392,14 @@ static inline int fuse_file_famfs(struct fuse_inode *fi)
 
 int fuse_get_fmap(struct fuse_mount *fm, struct inode *inode);
 
+int famfs_daxdev_open(struct fuse_conn *fc, struct fuse_backing_map *map);
+
 #else /* !CONFIG_FUSE_FAMFS_DAX */
 
+static inline void famfs_teardown(struct fuse_conn *fc)
+{
+}
+
 static inline struct fuse_backing *famfs_meta_set(struct fuse_inode *fi,
 						  void *meta)
 {
@@ -1392,6 +1410,10 @@ static inline void famfs_meta_free(struct fuse_inode *fi)
 {
 }
 
+static inline void famfs_init_devlist_sem(struct fuse_conn *fc)
+{
+}
+
 static inline int fuse_file_famfs(struct fuse_inode *fi)
 {
 	return 0;
@@ -1403,6 +1425,12 @@ fuse_get_fmap(struct fuse_mount *fm, struct inode *inode)
 	return 0;
 }
 
+static inline int
+famfs_daxdev_open(struct fuse_conn *fc, struct fuse_backing_map *map)
+{
+	return -EOPNOTSUPP;
+}
+
 #endif /* CONFIG_FUSE_FAMFS_DAX */
 
 #endif /* _FS_FUSE_I_H */
diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c
index 78ffc5fd50d0..9fc37015fb11 100644
--- a/fs/fuse/inode.c
+++ b/fs/fuse/inode.c
@@ -1021,6 +1021,9 @@ void fuse_conn_put(struct fuse_conn *fc)
 		WARN_ON(atomic_read(&bucket->count) != 1);
 		kfree(bucket);
 	}
+	if (IS_ENABLED(CONFIG_FUSE_FAMFS_DAX))
+		famfs_teardown(fc);
+
 	if (IS_ENABLED(CONFIG_FUSE_PASSTHROUGH))
 		fuse_backing_files_free(fc);
 	call_rcu(&fc->rcu, delayed_release);
@@ -1427,8 +1430,10 @@ static void process_init_reply(struct fuse_args *args, int error)
 				u64 in_flags = FIELD_PREP(GENMASK_ULL(63, 32), ia->in.flags2)
 						| ia->in.flags;
 
-				if (in_flags & FUSE_DAX_FMAP)
+				if (in_flags & FUSE_DAX_FMAP) {
+					famfs_init_devlist_sem(fc);
 					fc->famfs_iomap = 1;
+				}
 			}
 		} else {
 			ra_pages = fc->max_read / PAGE_SIZE;
diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h
index 4b84a58a8f1c..a143e6818416 100644
--- a/include/uapi/linux/fuse.h
+++ b/include/uapi/linux/fuse.h
@@ -1143,7 +1143,10 @@ struct fuse_notify_prune_out {
 struct fuse_backing_map {
 	int32_t		fd;
 	uint32_t	flags;
-	uint64_t	padding;
+	union {
+		uint64_t	padding;
+		uint64_t	daxdev_index;	/* FUSE_DEV_IOC_DAXDEV_OPEN */
+	};
 };
 
 /* Device ioctls: */
@@ -1153,6 +1156,8 @@ struct fuse_backing_map {
 					     struct fuse_backing_map)
 #define FUSE_DEV_IOC_BACKING_CLOSE	_IOW(FUSE_DEV_IOC_MAGIC, 2, uint32_t)
 #define FUSE_DEV_IOC_SYNC_INIT		_IO(FUSE_DEV_IOC_MAGIC, 3)
+#define FUSE_DEV_IOC_DAXDEV_OPEN	_IOW(FUSE_DEV_IOC_MAGIC, 4, \
+					     struct fuse_backing_map)
 
 struct fuse_lseek_in {
 	uint64_t	fh;
-- 
2.53.0
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.