[PATCH V11 4/9] famfs_fuse: Create files with famfs fmaps

John Groves <[email protected]> Mon, 20 Jul 2026 03:46:01 +0000
Newsgroups dev.linux.lists.nvdimm,dev.linux.lists.fuse-devel,org.kernel.vger.linux-cxl,org.kernel.vger.linux-doc,org.kernel.vger.linux-fsdevel,org.kernel.vger.linux-kernel
Message-ID <0100019f7da174eb-4cd0f3ae-59da-4533-b469-6eee009a22bd-000000@email.amazonses.com>
From: John Groves <[email protected]>

Interpret the fmap retrieved by GET_FMAP into the in-memory metadata that
later patches use to resolve read/write/mmap directly to dax.

- uapi: add the fmap wire format -- struct fuse_famfs_fmap_header and
  struct fuse_famfs_simple_ext, plus enum fuse_famfs_file_type and enum
  famfs_ext_type. An fmap is a simple (linear) list of extents, each
  mapping a file-offset range to a range on a dax device. A striped file
  is expressed by unrolling it into a longer simple-extent list in
  userspace, so the kernel format stays simple-only.
- famfs_kfmap.h: add the in-memory structures (struct famfs_file_meta,
  struct famfs_meta_simple_ext) that hang from fuse_inode->famfs_meta.
- famfs.c: add famfs_fuse_meta_alloc()/famfs_file_init_dax() to parse and
  validate the fmap and build the in-memory form, plus __famfs_meta_free().
  While parsing, detect a uniform (power-of-two) extent size and record it
  as meta->ext_shift, so the resolver can index the extent list by a shift
  (O(1)) instead of walking it; a non-uniform list falls back to a walk
  (ext_shift == 0).
- inode.c: only allow famfs mode if the fuse server holds CAP_SYS_RAWIO.
- Update MAINTAINERS for the new file.

The dev_dax_iomap plumbing that consumes this metadata is added in a
later patch.

Signed-off-by: John Groves <[email protected]>
---
 MAINTAINERS               |   1 +
 fs/fuse/famfs.c           | 240 +++++++++++++++++++++++++++++++++++++-
 fs/fuse/famfs_kfmap.h     |  64 ++++++++++
 fs/fuse/fuse_i.h          |   8 +-
 fs/fuse/inode.c           |  20 +++-
 include/uapi/linux/fuse.h |  40 +++++++
 6 files changed, 363 insertions(+), 10 deletions(-)
 create mode 100644 fs/fuse/famfs_kfmap.h

diff --git a/MAINTAINERS b/MAINTAINERS
index 0d0fded4fddb..07db3a5a2af7 100644
--- a/MAINTAINERS
+++ b/MAINTAINERS
@@ -10719,6 +10719,7 @@ L:	[email protected]
 L:	[email protected]
 S:	Supported
 F:	fs/fuse/famfs.c
+F:	fs/fuse/famfs_kfmap.h
 
 FUTEX SUBSYSTEM
 M:	Thomas Gleixner <[email protected]>
diff --git a/fs/fuse/famfs.c b/fs/fuse/famfs.c
index 80e6640ac970..8f7ee7d6151b 100644
--- a/fs/fuse/famfs.c
+++ b/fs/fuse/famfs.c
@@ -14,13 +14,244 @@
 #include <linux/mm.h>
 #include <linux/dax.h>
 #include <linux/iomap.h>
+#include <linux/log2.h>
 #include <linux/path.h>
 #include <linux/namei.h>
 #include <linux/string.h>
 
+#include "famfs_kfmap.h"
 #include "fuse_i.h"
 
 
+/***************************************************************************/
+
+void __famfs_meta_free(void *famfs_meta)
+{
+	struct famfs_file_meta *fmap = famfs_meta;
+
+	if (!fmap)
+		return;
+
+	kfree(fmap->se);
+	kfree(fmap);
+}
+DEFINE_FREE(__famfs_meta_free, void *, if (_T) __famfs_meta_free(_T))
+
+static int
+famfs_check_ext_alignment(struct famfs_meta_simple_ext *se)
+{
+	int errs = 0;
+
+	if (se->dev_index != 0)
+		errs++;
+
+	/* TODO: pass in alignment so we can support the other page sizes */
+	if (!IS_ALIGNED(se->ext_offset, PMD_SIZE))
+		errs++;
+
+	if (!IS_ALIGNED(se->ext_len, PMD_SIZE))
+		errs++;
+
+	return errs;
+}
+
+/**
+ * famfs_fuse_meta_alloc() - Allocate famfs file metadata
+ * @fmap_buf:  fmap buffer from fuse server
+ * @fmap_buf_size: size of fmap buffer
+ * @metap:         pointer where 'struct famfs_file_meta' is returned
+ *
+ * Returns: 0=success
+ *          -errno=failure
+ */
+static int
+famfs_fuse_meta_alloc(
+	void *fmap_buf,
+	size_t fmap_buf_size,
+	struct famfs_file_meta **metap)
+{
+	struct fuse_famfs_fmap_header *fmh;
+	size_t extent_total = 0;
+	size_t next_offset = 0;
+	int errs = 0;
+	int i;
+
+	fmh = fmap_buf;
+
+	/* Move past fmh in fmap_buf */
+	next_offset += sizeof(*fmh);
+	if (next_offset > fmap_buf_size) {
+		pr_err("%s:%d: fmap_buf underflow offset/size %ld/%ld\n",
+		       __func__, __LINE__, next_offset, fmap_buf_size);
+		return -EINVAL;
+	}
+
+	if (fmh->nextents < 1) {
+		pr_err("%s: nextents %d < 1\n", __func__, fmh->nextents);
+		return -ERANGE;
+	}
+
+	/*
+	 * No separate upper cap on nextents: the reply buffer bounds it. The
+	 * extent list is rejected below if it does not fit in fmap_buf_size, and
+	 * fuse_get_fmap() already refused to kvmalloc a buffer larger than
+	 * FMAP_BUFSIZE_MAX -- so anything that fits is small enough to handle.
+	 */
+
+	struct famfs_file_meta *meta __free(__famfs_meta_free) = kzalloc(sizeof(*meta), GFP_KERNEL);
+
+	if (!meta)
+		return -ENOMEM;
+
+	meta->error = false;
+	meta->file_type = fmh->file_type;
+	meta->file_size = fmh->file_size;
+
+	switch (fmh->ext_type) {
+	case FUSE_FAMFS_EXT_SIMPLE: {
+		struct fuse_famfs_simple_ext *se_in;
+
+		se_in = fmap_buf + next_offset;
+
+		/* Move past simple extents */
+		next_offset += fmh->nextents * sizeof(*se_in);
+		if (next_offset > fmap_buf_size) {
+			pr_err("%s:%d: fmap_buf underflow offset/size %ld/%ld\n",
+			       __func__, __LINE__, next_offset, fmap_buf_size);
+			return -EINVAL;
+		}
+
+		meta->fm_nextents = fmh->nextents;
+
+		meta->se = kcalloc(meta->fm_nextents, sizeof(*(meta->se)),
+				   GFP_KERNEL);
+		if (!meta->se)
+			return -ENOMEM;
+
+		for (i = 0; i < fmh->nextents; i++) {
+			meta->se[i].dev_index  = se_in[i].se_devindex;
+			meta->se[i].ext_offset = se_in[i].se_offset;
+			meta->se[i].ext_len    = se_in[i].se_len;
+
+			/* Record bitmap of referenced daxdev indices */
+			meta->dev_bitmap |= BIT_ULL(meta->se[i].dev_index);
+
+			errs += famfs_check_ext_alignment(&meta->se[i]);
+
+			extent_total += meta->se[i].ext_len;
+		}
+
+		/*
+		 * Detect a uniform extent size so the resolver can index se[]
+		 * by a shift rather than walking the list. This requires every
+		 * extent but the last to be the same power-of-2 size, with the
+		 * last no larger. ext_shift stays 0 (walk) for a single extent,
+		 * or a non-uniform / non-power-of-2 list.
+		 */
+		meta->ext_shift = 0;
+		if (meta->fm_nextents > 1) {
+			u64 esz = meta->se[0].ext_len;
+			bool uniform = is_power_of_2(esz);
+
+			for (i = 1; uniform && i < meta->fm_nextents - 1; i++)
+				if (meta->se[i].ext_len != esz)
+					uniform = false;
+
+			if (uniform && meta->se[meta->fm_nextents - 1].ext_len > esz)
+				uniform = false;
+
+			if (uniform)
+				meta->ext_shift = ilog2(esz);
+		}
+		break;
+	}
+
+	default:
+		pr_err("%s: invalid ext_type %d\n", __func__, fmh->ext_type);
+		return -EINVAL;
+	}
+
+	if (errs > 0) {
+		pr_err("%s: %d alignment errors found\n", __func__, errs);
+		return -EINVAL;
+	}
+
+	/* More sanity checks */
+	if (extent_total < meta->file_size) {
+		pr_err("%s: file size %ld larger than map size %ld\n",
+		       __func__, meta->file_size, extent_total);
+		return -EINVAL;
+	}
+
+	if (cmpxchg(metap, NULL, meta) != NULL) {
+		pr_debug("%s: fmap race detected\n", __func__);
+		return 0; /* fmap already installed */
+	}
+	retain_and_null_ptr(meta);
+
+	return 0;
+}
+
+/**
+ * famfs_file_init_dax() - init famfs dax file metadata
+ *
+ * @fm:        fuse_mount
+ * @inode:     the inode
+ * @fmap_buf:  fmap response message
+ * @fmap_size: Size of the fmap message
+ *
+ * Initialize famfs metadata for a file, based on the contents of the GET_FMAP
+ * response
+ *
+ * Return: 0=success
+ *          -errno=failure
+ */
+int
+famfs_file_init_dax(
+	struct fuse_mount *fm,
+	struct inode *inode,
+	void *fmap_buf,
+	size_t fmap_size)
+{
+	struct fuse_inode *fi = get_fuse_inode(inode);
+	struct famfs_file_meta *meta = NULL;
+	int rc;
+
+	if (fi->famfs_meta) {
+		pr_notice("%s: i_no=%llu fmap_size=%zu ALREADY INITIALIZED\n",
+			  __func__,
+			  inode->i_ino, fmap_size);
+		return 0;
+	}
+
+	rc = famfs_fuse_meta_alloc(fmap_buf, fmap_size, &meta);
+	if (rc)
+		goto errout;
+
+	/* Publish the famfs metadata on fi->famfs_meta */
+	inode_lock(inode);
+
+	if (famfs_meta_set(fi, meta) == NULL) {
+		i_size_write(inode, meta->file_size);
+		inode->i_flags |= S_DAX;
+	} else {
+		pr_debug("%s: file already had metadata\n", __func__);
+		__famfs_meta_free(meta);
+		/* rc is 0 - the file is valid */
+	}
+
+	inode_unlock(inode);
+	return 0;
+
+errout:
+	if (rc)
+		__famfs_meta_free(meta);
+
+	return rc;
+}
+
+#define FMAP_BUFSIZE PAGE_SIZE
+
 #define FMAP_BUFSIZE_INIT PAGE_SIZE
 /*
  * Largest GET_FMAP reply buffer we will kvmalloc. Any fmap whose whole message
@@ -123,12 +354,9 @@ int fuse_get_fmap(struct fuse_mount *fm, struct inode *inode)
 		bufsize = required;
 	}
 
-	/* We retrieved the "fmap" (the file's map to memory), but
-	 * we haven't used it yet. A call to famfs_file_init_dax() will be added
-	 * here in a subsequent patch, when we add the ability to attach
-	 * fmaps to files.
-	 */
+	/* Convert fmap into in-memory format and hang from inode */
+	rc = famfs_file_init_dax(fm, inode, fmap_buf, fmap_size);
 
 	kvfree(fmap_buf);
-	return 0;
+	return rc;
 }
diff --git a/fs/fuse/famfs_kfmap.h b/fs/fuse/famfs_kfmap.h
new file mode 100644
index 000000000000..d87b065e8ac8
--- /dev/null
+++ b/fs/fuse/famfs_kfmap.h
@@ -0,0 +1,64 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * famfs - dax file system for shared fabric-attached memory
+ *
+ * Copyright 2023-2026 Micron Technology, Inc.
+ */
+#ifndef FAMFS_KFMAP_H
+#define FAMFS_KFMAP_H
+
+/* KABI version 43 (aka v2) fmap structures
+ *
+ * The location of the memory backing for a famfs file is described by
+ * the response to the GET_FMAP fuse message (defined in
+ * include/uapi/linux/fuse.h).
+ *
+ * A famfs file is described by a list of simple extents: (devindex, offset,
+ * length) tuples, where devindex references a devdax device that has been
+ * registered with the kernel. The extent list must cover at least file_size.
+ * Striping/interleaving is expressed by unrolling into a longer simple-extent
+ * list in userspace; the kernel handles only simple extents.
+ */
+
+/*
+ * The structures below are the in-memory metadata format for famfs files.
+ * Metadata retrieved via the GET_FMAP response is converted to this format
+ * for use in resolving file mapping faults.
+ *
+ * The GET_FMAP response contains the same information, but in a more
+ * message-and-versioning-friendly format. Those structs can be found in the
+ * famfs section of include/uapi/linux/fuse.h (aka fuse_kernel.h in libfuse)
+ */
+
+enum famfs_file_type {
+	FAMFS_REG,
+	FAMFS_SUPERBLOCK,
+	FAMFS_LOG,
+};
+
+struct famfs_meta_simple_ext {
+	u64 dev_index;
+	u64 ext_offset;
+	u64 ext_len;
+};
+
+/*
+ * Each famfs dax file has this hanging from its fuse_inode->famfs_meta
+ */
+struct famfs_file_meta {
+	bool                   error;
+	enum famfs_file_type   file_type;
+	size_t                 file_size;
+	u64 dev_bitmap; /* bitmap of referenced daxdevs by index */
+	size_t                 fm_nextents;
+	/*
+	 * If every extent but the last is the same power-of-2 size, ext_shift
+	 * is ilog2(that size) and the resolver indexes se[] by a shift of the
+	 * file offset (O(1)). Otherwise ext_shift is 0 and the resolver walks
+	 * the list (single-extent or non-uniform files).
+	 */
+	u32                    ext_shift;
+	struct famfs_meta_simple_ext  *se;
+};
+
+#endif /* FAMFS_KFMAP_H */
diff --git a/fs/fuse/fuse_i.h b/fs/fuse/fuse_i.h
index 5bacc5098620..46a7040b38dc 100644
--- a/fs/fuse/fuse_i.h
+++ b/fs/fuse/fuse_i.h
@@ -1347,6 +1347,9 @@ extern void fuse_sysctl_unregister(void);
 /* famfs.c */
 
 #if IS_ENABLED(CONFIG_FUSE_FAMFS_DAX)
+int famfs_file_init_dax(struct fuse_mount *fm,
+			struct inode *inode, void *fmap_buf,
+			size_t fmap_size);
 void __famfs_meta_free(void *map);
 
 /* Set fi->famfs_meta = NULL regardless of prior value */
@@ -1364,7 +1367,10 @@ static inline struct fuse_backing *famfs_meta_set(struct fuse_inode *fi,
 
 static inline void famfs_meta_free(struct fuse_inode *fi)
 {
-	famfs_meta_set(fi, NULL);
+	if (fi->famfs_meta != NULL) {
+		__famfs_meta_free(fi->famfs_meta);
+		WRITE_ONCE(fi->famfs_meta, NULL);
+	}
 }
 
 static inline int fuse_file_famfs(struct fuse_inode *fi)
diff --git a/fs/fuse/inode.c b/fs/fuse/inode.c
index e030a302120f..78ffc5fd50d0 100644
--- a/fs/fuse/inode.c
+++ b/fs/fuse/inode.c
@@ -7,6 +7,7 @@
 #include "dev.h"
 #include "fuse_i.h"
 
+#include <linux/bitfield.h>
 #include <linux/dax.h>
 #include <linux/pagemap.h>
 #include <linux/slab.h>
@@ -1414,8 +1415,21 @@ static void process_init_reply(struct fuse_args *args, int error)
 				timeout = arg->request_timeout;
 
 			if (IS_ENABLED(CONFIG_FUSE_FAMFS_DAX) &&
-			    flags & FUSE_DAX_FMAP)
-				fc->famfs_iomap = 1;
+			    flags & FUSE_DAX_FMAP) {
+				/* famfs_iomap is only allowed if the fuse
+				 * server has CAP_SYS_RAWIO. This was checked
+				 * in fuse_send_init, and FUSE_DAX_IOMAP was
+				 * set in in_flags if so. Only allow enablement
+				 * if we find it there. This function is
+				 * normally not running in fuse server context,
+				 * so we can't do the capability check here...
+				 */
+				u64 in_flags = FIELD_PREP(GENMASK_ULL(63, 32), ia->in.flags2)
+						| ia->in.flags;
+
+				if (in_flags & FUSE_DAX_FMAP)
+					fc->famfs_iomap = 1;
+			}
 		} else {
 			ra_pages = fc->max_read / PAGE_SIZE;
 			fc->no_lock = 1;
@@ -1483,7 +1497,7 @@ static struct fuse_init_args *fuse_new_init(struct fuse_mount *fm)
 		flags |= FUSE_SUBMOUNTS;
 	if (IS_ENABLED(CONFIG_FUSE_PASSTHROUGH))
 		flags |= FUSE_PASSTHROUGH;
-	if (IS_ENABLED(CONFIG_FUSE_FAMFS_DAX))
+	if (IS_ENABLED(CONFIG_FUSE_FAMFS_DAX) && capable(CAP_SYS_RAWIO))
 		flags |= FUSE_DAX_FMAP;
 
 	/*
diff --git a/include/uapi/linux/fuse.h b/include/uapi/linux/fuse.h
index d323c20e79bd..4b84a58a8f1c 100644
--- a/include/uapi/linux/fuse.h
+++ b/include/uapi/linux/fuse.h
@@ -243,6 +243,12 @@
  *
  *  7.46
  *  - Add FUSE_DAX_FMAP capability - ability to handle in-kernel fsdax maps
+ *  - Add the following structures for the GET_FMAP message reply components:
+ *    - struct fuse_famfs_simple_ext
+ *    - struct fuse_famfs_fmap_header
+ *  - Add the following enumerated types
+ *    - enum fuse_famfs_file_type
+ *    - enum famfs_ext_type
  */
 
 #ifndef _LINUX_FUSE_H
@@ -1316,4 +1322,38 @@ struct fuse_uring_cmd_req {
 	uint8_t padding[6];
 };
 
+/* Famfs fmap message components */
+
+#define FAMFS_FMAP_VERSION 1
+
+#define FAMFS_FMAP_MAX 32768 /* Largest supported fmap message */
+
+enum fuse_famfs_file_type {
+	FUSE_FAMFS_FILE_REG,
+	FUSE_FAMFS_FILE_SUPERBLOCK,
+	FUSE_FAMFS_FILE_LOG,
+};
+
+enum famfs_ext_type {
+	FUSE_FAMFS_EXT_SIMPLE = 0,
+};
+
+struct fuse_famfs_simple_ext {
+	uint32_t se_devindex;
+	uint32_t reserved;
+	uint64_t se_offset;
+	uint64_t se_len;
+};
+
+struct fuse_famfs_fmap_header {
+	uint8_t file_type; /* enum fuse_famfs_file_type */
+	uint8_t reserved;
+	uint16_t fmap_version;
+	uint32_t ext_type; /* enum famfs_ext_type */
+	uint32_t nextents;
+	uint32_t fmap_size; /* Inclusive of this header */
+	uint64_t file_size;
+	uint64_t reserved1;
+};
+
 #endif /* _LINUX_FUSE_H */
-- 
2.53.0