[PATCH V11 5/9] famfs_fuse: register fs-dax daxdevs via FUSE_DEV_IOC_DAXDEV_OPEN

John Groves <[email protected]> Mon, 20 Jul 2026 03:46:13 +0000
Newsgroups dev.linux.lists.nvdimm,dev.linux.lists.fuse-devel,org.kernel.vger.linux-cxl,org.kernel.vger.linux-doc,org.kernel.vger.linux-fsdevel,org.kernel.vger.linux-kernel
Message-ID <0100019f7da1a534-2fc6658c-4f30-4cc6-b010-5e3f8fd7164b-000000@email.amazonses.com>
From: John Groves <[email protected]>

Add a dedicated ioctl, FUSE_DEV_IOC_DAXDEV_OPEN, by which a fuse server
registers the devdax devices that back an fs-dax (famfs) filesystem: the
server hands the kernel an fd to a /dev/daxN.N device plus its
cluster-invariant famfs index.

A dedicated ioctl, rather than overloading FUSE_DEV_IOC_BACKING_OPEN,
avoids a dependency on CONFIG_FUSE_PASSTHROUGH and the backing-file
machinery, which famfs does not use.

- struct fuse_backing_map is reused as the argument; the index rides on
  the reserved 'padding' field (daxdev_index).
- fuse_dev_ioctl_daxdev_open() copies the map and calls
  famfs_daxdev_open(), which resolves the fd to a dax device by its inode
  i_rdev. The daxdev table store is added in the following patch.

Access control:

- The ioctl is gated on famfs mode (enabled at FUSE_INIT only when the
  server held CAP_SYS_RAWIO), and additionally re-checks
  capable(CAP_SYS_RAWIO) on each call. The famfs-mode flag attests only
  to the session founder's privilege; the fuse device fd may be inherited
  across fork() or SCM_RIGHTS-passed to a less-privileged task, so the
  capability is re-checked against the task actually performing the
  registration.
- famfs_daxdev_open() resolves the fd with fget(), not fget_raw(), so
  O_PATH fds are rejected. A successful lookup then proves the caller
  holds an fd from a real open() of the daxdev -- i.e. it already passed
  may_open_dev() and the device node's DAC checks -- rather than an
  O_PATH reference that bypasses them.

Signed-off-by: John Groves <[email protected]>
---
 fs/fuse/dev.c             |  31 ++++
 fs/fuse/dev.h             |   1 +
 fs/fuse/famfs.c           | 330 ++++++++++++++++++++++++++++++++++++++
 fs/fuse/famfs_kfmap.h     |  27 ++++
 fs/fuse/fuse_i.h          |  28 ++++
 fs/fuse/inode.c           |   7 +-
 include/uapi/linux/fuse.h |   7 +-
 7 files changed, 429 insertions(+), 2 deletions(-)

diff --git a/fs/fuse/dev.c b/fs/fuse/dev.c
index 5763a7cd3b37..3e6aa15e0346 100644
--- a/fs/fuse/dev.c
+++ b/fs/fuse/dev.c
@@ -2330,6 +2330,34 @@ static long fuse_dev_ioctl_backing_close(struct file *file, __u32 __user *argp)
 	return fuse_backing_close(fud->chan->conn, backing_id);
 }
 
+static long fuse_dev_ioctl_daxdev_open(struct file *file,
+				       struct fuse_backing_map __user *argp)
+{
+	struct fuse_dev *fud = fuse_get_dev(file);
+	struct fuse_backing_map map;
+
+	if (IS_ERR(fud))
+		return PTR_ERR(fud);
+
+	if (!IS_ENABLED(CONFIG_FUSE_FAMFS_DAX))
+		return -EOPNOTSUPP;
+
+	/*
+	 * The famfs-mode gate (fc->famfs_iomap) lives in famfs_daxdev_open(),
+	 * which has the full fuse_conn definition. Here, re-check CAP_SYS_RAWIO
+	 * against the task performing the registration: famfs mode being enabled
+	 * only attests that the session founder held it at FUSE_INIT, and the
+	 * fuse device fd may have been passed to a less-privileged process.
+	 */
+	if (!capable(CAP_SYS_RAWIO))
+		return -EPERM;
+
+	if (copy_from_user(&map, argp, sizeof(map)))
+		return -EFAULT;
+
+	return famfs_daxdev_open(fud->chan->conn, &map);
+}
+
 static long fuse_dev_ioctl_sync_init(struct file *file)
 {
 	struct fuse_dev *fud = fuse_file_to_fud(file);
@@ -2359,6 +2387,9 @@ static long fuse_dev_ioctl(struct file *file, unsigned int cmd,
 	case FUSE_DEV_IOC_SYNC_INIT:
 		return fuse_dev_ioctl_sync_init(file);
 
+	case FUSE_DEV_IOC_DAXDEV_OPEN:
+		return fuse_dev_ioctl_daxdev_open(file, argp);
+
 	default:
 		return -ENOTTY;
 	}
diff --git a/fs/fuse/dev.h b/fs/fuse/dev.h
index aed69fd14c41..545940b635cc 100644
--- a/fs/fuse/dev.h
+++ b/fs/fuse/dev.h
@@ -87,6 +87,7 @@ int fuse_notify(struct fuse_conn *fc, enum fuse_notify_code code,
 
 int fuse_backing_open(struct fuse_conn *fc, struct fuse_backing_map *map);
 int fuse_backing_close(struct fuse_conn *fc, int backing_id);
+int famfs_daxdev_open(struct fuse_conn *fc, struct fuse_backing_map *map);
 
 int fuse_copy_one(struct fuse_copy_state *cs, void *val, unsigned size);
 int fuse_copy_folio(struct fuse_copy_state *cs, struct folio **foliop,
diff --git a/fs/fuse/famfs.c b/fs/fuse/famfs.c
index 8f7ee7d6151b..a2a7dd631dc0 100644
--- a/fs/fuse/famfs.c
+++ b/fs/fuse/famfs.c
@@ -11,6 +11,7 @@
 
 #include <linux/cleanup.h>
 #include <linux/fs.h>
+#include <linux/file.h>
 #include <linux/mm.h>
 #include <linux/dax.h>
 #include <linux/iomap.h>
@@ -22,6 +23,331 @@
 #include "famfs_kfmap.h"
 #include "fuse_i.h"
 
+static void famfs_set_daxdev_err(
+	struct fuse_conn *fc, struct dax_device *dax_devp);
+
+static int
+famfs_dax_notify_failure(struct dax_device *dax_devp, u64 offset,
+			u64 len, int mf_flags)
+{
+	struct fuse_conn *fc = dax_holder(dax_devp);
+
+	famfs_set_daxdev_err(fc, dax_devp);
+
+	return 0;
+}
+
+static const struct dax_holder_operations famfs_fuse_dax_holder_ops = {
+	.notify_failure		= famfs_dax_notify_failure,
+};
+
+/*****************************************************************************/
+
+/*
+ * famfs_teardown()
+ *
+ * Deallocate famfs metadata for a fuse_conn
+ */
+void
+famfs_teardown(struct fuse_conn *fc)
+{
+	struct famfs_dax_devlist *devlist __free(kfree) = NULL;
+	int i;
+
+	/*
+	 * Detach the table under the same lock famfs_set_daxdev_err() takes, so
+	 * a notify_failure racing teardown either runs first against the live
+	 * table or observes dax_devlist == NULL and bails, rather than
+	 * dereferencing it after we clear it. The daxdev holders are dropped
+	 * below, after which no further notify_failure can arrive.
+	 */
+	scoped_guard(rwsem_write, &fc->famfs_devlist_sem) {
+		devlist = fc->dax_devlist;
+		fc->dax_devlist = NULL;
+	}
+
+	if (!devlist)
+		return;
+
+	if (!devlist->devlist)
+		return;
+
+	/* Close & release all the daxdevs in our table */
+	for (i = 0; i < devlist->nslots; i++) {
+		struct famfs_daxdev *dd = &devlist->devlist[i];
+
+		if (!dd->valid)
+			continue;
+
+		/* Only call fs_put_dax if fs_dax_get succeeded */
+		if (dd->devp) {
+			if (!dd->dax_err)
+				fs_put_dax(dd->devp, fc);
+			put_dax(dd->devp);
+		}
+
+		kfree(dd->name);
+	}
+	kfree(devlist->devlist);
+}
+
+/* Allocate the daxdev table on first use (idempotent via cmpxchg) */
+static int famfs_devlist_alloc(struct fuse_conn *fc)
+{
+	struct famfs_dax_devlist *devlist;
+
+	if (fc->dax_devlist)
+		return 0;
+
+	devlist = kcalloc(1, sizeof(*devlist), GFP_KERNEL);
+	if (!devlist)
+		return -ENOMEM;
+
+	devlist->nslots = MAX_DAXDEVS;
+	devlist->devlist = kcalloc(MAX_DAXDEVS, sizeof(struct famfs_daxdev),
+				   GFP_KERNEL);
+	if (!devlist->devlist) {
+		kfree(devlist);
+		return -ENOMEM;
+	}
+
+	/* If another thread allocated it first, drop ours */
+	if (cmpxchg(&fc->dax_devlist, NULL, devlist) != NULL) {
+		kfree(devlist->devlist);
+		kfree(devlist);
+	}
+
+	return 0;
+}
+
+/*
+ * famfs_install_daxdev() - exclusively acquire a resolved daxdev and publish
+ * it in the table at @index. Shared by the GET_DAXDEV (pull) and
+ * DAXDEV_OPEN (push) registration paths.
+ *
+ * Serializes with concurrent installers under famfs_devlist_sem and rechecks
+ * ->valid. A daxdev is entered in the table only once it has been exclusively
+ * acquired via fs_dax_get(); on failure the dax_dev_get() reference is
+ * released and the slot is left invalid, so the referencing fmap is rejected
+ * rather than mapped without an exclusive holder. @name may be NULL (the push
+ * path passes no pathname).
+ */
+static int famfs_install_daxdev(struct fuse_conn *fc, u64 index, dev_t devno,
+				const char *name)
+{
+	struct famfs_daxdev *daxdev;
+	int rc = 0;
+
+	if (index >= fc->dax_devlist->nslots) {
+		pr_err("%s: index(%llu) >= nslots(%d)\n",
+		       __func__, index, fc->dax_devlist->nslots);
+		return -EINVAL;
+	}
+
+	scoped_guard(rwsem_write, &fc->famfs_devlist_sem) {
+		daxdev = &fc->dax_devlist->devlist[index];
+
+		/* Installed already by a concurrent push/pull */
+		if (daxdev->valid)
+			return 0;
+
+		/*
+		 * A prior attempt already determined this daxdev cannot be
+		 * exclusively acquired (see the fs_dax_get() failure handling
+		 * below). Don't thrash on GET_DAXDEV/fs_dax_get(); fail fast.
+		 */
+		if (daxdev->dax_err)
+			return -EIO;
+
+		/*
+		 * Temporary: dax_dev_get() is the exported upstream lookup, but
+		 * unlike dax_dev_find() it allocates for any dev_t and does not
+		 * reject non-dax devices. Restore dax_dev_find() (and that
+		 * rejection) once it is upstream.
+		 */
+		daxdev->devp = dax_dev_get(devno);
+		if (!daxdev->devp) {
+			pr_warn("%s: device %u:%u not found or not dax\n",
+				__func__, MAJOR(devno), MINOR(devno));
+			return -ENODEV;
+		}
+
+		rc = fs_dax_get(daxdev->devp, fc, &famfs_fuse_dax_holder_ops);
+		if (rc) {
+			/*
+			 * Distinguish a lost race from a real failure. -EBUSY
+			 * with the daxdev already held by *this* fuse_conn
+			 * means a concurrent acquire won and will publish the
+			 * slot valid: not an error, and must not be cached as
+			 * dax_err. Any other failure (foreign holder, not a dax
+			 * device, wrong driver type) is permanent for this
+			 * connection, so record dax_err to stop re-fetching and
+			 * re-acquiring it.
+			 */
+			if (!(rc == -EBUSY && dax_holder(daxdev->devp) == fc)) {
+				pr_err("%s: fs_dax_get(%u:%u) failed rc=%d\n",
+				       __func__, MAJOR(devno), MINOR(devno), rc);
+				daxdev->dax_err = true;
+			}
+			put_dax(daxdev->devp);
+			daxdev->devp = NULL;
+			return rc;
+		}
+
+		daxdev->devno = devno;
+		if (name) {
+			daxdev->name = kstrdup(name, GFP_KERNEL);
+			if (!daxdev->name) {
+				fs_put_dax(daxdev->devp, fc);
+				put_dax(daxdev->devp);
+				daxdev->devp = NULL;
+				return -ENOMEM;
+			}
+		}
+
+		wmb(); /* All other fields must be visible before valid */
+		daxdev->valid = 1;
+	}
+
+	return 0;
+}
+
+/**
+ * famfs_daxdev_open() - Register a daxdev via FUSE_DEV_IOC_DAXDEV_OPEN
+ * @fc:   fuse_conn
+ * @map:  fuse_backing_map; @map->fd is an fd to the devdax device and
+ *        @map->daxdev_index is the (cluster-invariant) famfs index.
+ *
+ * The server pushes a daxdev to the kernel by reference (an fd), rather than
+ * the kernel pulling it by name via GET_DAXDEV. The resolved daxdev is
+ * exclusively acquired and entered in the table at @map->daxdev_index.
+ *
+ * Return: 0=success
+ *         -errno=failure
+ */
+int famfs_daxdev_open(struct fuse_conn *fc, struct fuse_backing_map *map)
+{
+	struct inode *inode;
+	struct file *file;
+	dev_t devno;
+	int rc;
+
+	/* Only fs-dax (famfs) mode accepts daxdev registration */
+	if (!fc->famfs_iomap)
+		return -EOPNOTSUPP;
+
+	file = fget(map->fd);
+	if (!file)
+		return -EBADF;
+
+	inode = file_inode(file);
+	if (!S_ISCHR(inode->i_mode)) {
+		fput(file);
+		return -EINVAL;
+	}
+	devno = inode->i_rdev;
+	fput(file);
+
+	rc = famfs_devlist_alloc(fc);
+	if (rc)
+		return rc;
+
+	rc = famfs_install_daxdev(fc, map->daxdev_index, devno, NULL);
+	if (rc)
+		pr_err("%s: failed to install daxdev\n", __func__);
+
+	return rc;
+}
+
+/**
+ * famfs_check_daxdev_table() - Verify an fmap's referenced daxdevs are installed
+ * @fm:   fuse_mount
+ * @meta: famfs_file_meta, in-memory format, built from a GET_FMAP response
+ *
+ * Called for each new file fmap. Every daxdev the fmap references must already
+ * be installed in the table, having been pushed in via FUSE_DEV_IOC_DAXDEV_OPEN
+ * before any file that uses it is accessed. If any referenced daxdev is not
+ * present, the fmap is rejected so the file is never mapped against a daxdev
+ * that has no exclusive holder.
+ *
+ * Return: 0=success (all referenced daxdevs present)
+ *         <0=a referenced daxdev is missing from the table
+ */
+static int
+famfs_check_daxdev_table(
+	struct fuse_mount *fm,
+	const struct famfs_file_meta *meta)
+{
+	struct fuse_conn *fc = fm->fc;
+	int nmissing = 0;
+	int err;
+
+	err = famfs_devlist_alloc(fc);
+	if (err)
+		return err;
+
+	/* Count missing daxdevs while holding the reader lock */
+	scoped_guard(rwsem_read, &fc->famfs_devlist_sem) {
+		unsigned long i;
+
+		for_each_set_bit(i, (unsigned long *)&meta->dev_bitmap,
+				 MAX_DAXDEVS) {
+			struct famfs_daxdev *dd = &fc->dax_devlist->devlist[i];
+
+			/*
+			 * Skip daxdevs already installed (valid) or already
+			 * known to be unusable (dax_err). Re-fetching either
+			 * just thrashes on GET_DAXDEV and fs_dax_get().
+			 */
+			if (!dd->valid && !dd->dax_err)
+				nmissing++;
+		}
+	}
+
+	if (nmissing > 0) {
+		/* this file referenced at least one daxdev that is not in
+		 * the table. Daxdevs must be known before any file that
+		 * uses them is accessed
+		 */
+		pr_err("%s: %d missing daxdev(s)\n", __func__, nmissing);
+		return -ENODEV;
+	}
+
+	return 0;
+}
+
+static void
+famfs_set_daxdev_err(
+	struct fuse_conn *fc,
+	struct dax_device *dax_devp)
+{
+	int i;
+
+	/*
+	 * Search the list by dax_devp under the write lock: we set dd->error,
+	 * and it serializes against famfs_teardown() clearing the table.
+	 */
+	scoped_guard(rwsem_write, &fc->famfs_devlist_sem) {
+		if (!fc->dax_devlist)
+			return;
+		for (i = 0; i < fc->dax_devlist->nslots; i++) {
+			if (fc->dax_devlist->devlist[i].valid) {
+				struct famfs_daxdev *dd;
+
+				dd = &fc->dax_devlist->devlist[i];
+				if (dd->devp != dax_devp)
+					continue;
+
+				dd->error = true;
+
+				pr_err("%s: memory error on daxdev %s (%d)\n",
+				       __func__, dd->name, i);
+				return;
+			}
+		}
+	}
+	pr_err("%s: memory err on unrecognized daxdev\n", __func__);
+}
 
 /***************************************************************************/
 
@@ -228,6 +554,10 @@ famfs_file_init_dax(
 	if (rc)
 		goto errout;
 
+	/* Make sure this fmap doesn't reference any unknown daxdevs */
+	if (famfs_check_daxdev_table(fm, meta))
+		meta->error = true;
+
 	/* Publish the famfs metadata on fi->famfs_meta */
 	inode_lock(inode);
 
diff --git a/fs/fuse/famfs_kfmap.h b/fs/fuse/famfs_kfmap.h
index d87b065e8ac8..6b78e78325c8 100644
--- a/fs/fuse/famfs_kfmap.h
+++ b/fs/fuse/famfs_kfmap.h
@@ -61,4 +61,31 @@ struct famfs_file_meta {
 	struct famfs_meta_simple_ext  *se;
 };
 
+/*
+ * famfs_daxdev - tracking struct for a daxdev within a famfs file system
+ *
+ * This is the in-memory daxdev metadata that is populated by parsing
+ * the responses to GET_FMAP messages
+ */
+struct famfs_daxdev {
+	/* Include dev uuid? */
+	bool valid;
+	bool error; /* Dax has reported a memory error (probably poison) */
+	bool dax_err; /* fs_dax_get() failed */
+	dev_t devno;
+	struct dax_device *devp;
+	char *name;
+};
+
+#define MAX_DAXDEVS 24
+
+/*
+ * famfs_dax_devlist - list of famfs_daxdev's
+ */
+struct famfs_dax_devlist {
+	int nslots;
+	int ndevs;
+	struct famfs_daxdev *devlist;
+};
+
 #endif /* FAMFS_KFMAP_H */
diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h
index 46a7040b38dc..5394aae9dbac 100644
--- a/fs/fuse/fuse_i.h
+++ b/fs/fuse/fuse_i.h
@@ -773,6 +773,11 @@ struct fuse_conn {
 	/** @backing_files_map: IDR for backing files ids */
 	struct idr backing_files_map;
 #endif
+
+#if IS_ENABLED(CONFIG_FUSE_FAMFS_DAX)
+	struct rw_semaphore famfs_devlist_sem;
+	struct famfs_dax_devlist *dax_devlist;
+#endif
 };
 
 /*
@@ -1352,6 +1357,8 @@ int famfs_file_init_dax(struct fuse_mount *fm,
 			size_t fmap_size);
 void __famfs_meta_free(void *map);
 
+void famfs_teardown(struct fuse_conn *fc);
+
 /* Set fi->famfs_meta = NULL regardless of prior value */
 static inline void famfs_meta_init(struct fuse_inode *fi)
 {
@@ -1373,6 +1380,11 @@ static inline void famfs_meta_free(struct fuse_inode *fi)
 	}
 }
 
+static inline void famfs_init_devlist_sem(struct fuse_conn *fc)
+{
+	init_rwsem(&fc->famfs_devlist_sem);
+}
+
 static inline int fuse_file_famfs(struct fuse_inode *fi)
 {
 	return (READ_ONCE(fi->famfs_meta) != NULL);
@@ -1380,8 +1392,14 @@ static inline int fuse_file_famfs(struct fuse_inode *fi)
 
 int fuse_get_fmap(struct fuse_mount *fm, struct inode *inode);
 
+int famfs_daxdev_open(struct fuse_conn *fc, struct fuse_backing_map *map);
+
 #else /* !CONFIG_FUSE_FAMFS_DAX */
 
+static inline void famfs_teardown(struct fuse_conn *fc)
+{
+}
+
 static inline struct fuse_backing *famfs_meta_set(struct fuse_inode *fi,
 						  void *meta)
 {
@@ -1392,6 +1410,10 @@ static inline void famfs_meta_free(struct fuse_inode *fi)
 {
 }
 
+static inline void famfs_init_devlist_sem(struct fuse_conn *fc)
+{
+}
+
 static inline int fuse_file_famfs(struct fuse_inode *fi)
 {
 	return 0;
@@ -1403,6 +1425,12 @@ fuse_get_fmap(struct fuse_mount *fm, struct inode *inode)
 	return 0;
 }
 
+static inline int
+famfs_daxdev_open(struct fuse_conn *fc, struct fuse_backing_map *map)
+{
+	return -EOPNOTSUPP;
+}
+
 #endif /* CONFIG_FUSE_FAMFS_DAX */
 
 #endif /* _FS_FUSE_I_H */
diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c
index 78ffc5fd50d0..9fc37015fb11 100644
--- a/fs/fuse/inode.c
+++ b/fs/fuse/inode.c
@@ -1021,6 +1021,9 @@ void fuse_conn_put(struct fuse_conn *fc)
 		WARN_ON(atomic_read(&bucket->count) != 1);
 		kfree(bucket);
 	}
+	if (IS_ENABLED(CONFIG_FUSE_FAMFS_DAX))
+		famfs_teardown(fc);
+
 	if (IS_ENABLED(CONFIG_FUSE_PASSTHROUGH))
 		fuse_backing_files_free(fc);
 	call_rcu(&fc->rcu, delayed_release);
@@ -1427,8 +1430,10 @@ static void process_init_reply(struct fuse_args *args, int error)
 				u64 in_flags = FIELD_PREP(GENMASK_ULL(63, 32), ia->in.flags2)
 						| ia->in.flags;
 
-				if (in_flags & FUSE_DAX_FMAP)
+				if (in_flags & FUSE_DAX_FMAP) {
+					famfs_init_devlist_sem(fc);
 					fc->famfs_iomap = 1;
+				}
 			}
 		} else {
 			ra_pages = fc->max_read / PAGE_SIZE;
diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h
index 4b84a58a8f1c..a143e6818416 100644
--- a/include/uapi/linux/fuse.h
+++ b/include/uapi/linux/fuse.h
@@ -1143,7 +1143,10 @@ struct fuse_notify_prune_out {
 struct fuse_backing_map {
 	int32_t		fd;
 	uint32_t	flags;
-	uint64_t	padding;
+	union {
+		uint64_t	padding;
+		uint64_t	daxdev_index;	/* FUSE_DEV_IOC_DAXDEV_OPEN */
+	};
 };
 
 /* Device ioctls: */
@@ -1153,6 +1156,8 @@ struct fuse_backing_map {
 					     struct fuse_backing_map)
 #define FUSE_DEV_IOC_BACKING_CLOSE	_IOW(FUSE_DEV_IOC_MAGIC, 2, uint32_t)
 #define FUSE_DEV_IOC_SYNC_INIT		_IO(FUSE_DEV_IOC_MAGIC, 3)
+#define FUSE_DEV_IOC_DAXDEV_OPEN	_IOW(FUSE_DEV_IOC_MAGIC, 4, \
+					     struct fuse_backing_map)
 
 struct fuse_lseek_in {
 	uint64_t	fh;
-- 
2.53.0