Re: [PATCH] erofs: reuse superblock for file-backed mounts

Giuseppe Scrivano <[email protected]>
Newsgroups org.kernel.vger.linux-fsdevel,org.ozlabs.lists.linux-erofs
Message-ID <[email protected]>
Giuseppe Scrivano <[email protected]> writes:

> Pedro Falcato <[email protected]> writes:
>
>> On Thu, Jul 30, 2026 at 02:41:20PM +0200, Giuseppe Scrivano wrote:
>>> When the same file is mounted multiple times (via path or fd), reuse
>>> the existing superblock instead of creating a new one.  This allows
>>> multiple mounts of the same image to share in-kernel data structures
>>> more efficiently.
>>> 
>>> The backing file is identified by its inode and fsoffset.  If mount
>>> options conflict, a separate superblock is created transparently as
>>> a fallback.
>>> 
>>> Tested by mounting a 177M Fedora EROFS image 20 times with full
>>> traversal:
>>> 
>>> ```
>>> \#!/bin/sh
>>> IMG=${1:-/root/fedora.erofs}
>>> N=20
>>> DIR=$(mktemp -d)
>>> trap "umount $DIR/m* 2>/dev/null; rm -rf $DIR" EXIT
>>> 
>>> sync; echo 3 > /proc/sys/vm/drop_caches
>>> INODES_BEFORE=$(grep erofs_inode /proc/slabinfo | awk '{print $2}')
>>> MEM_BEFORE=$(grep ^Slab: /proc/meminfo | awk '{print $2}')
>>> 
>>> for i in $(seq 1 $N); do mkdir $DIR/m$i && mount -t erofs "$IMG"
>>> $DIR/m$i && find $DIR/m$i > /dev/null; done
>>> 
>>> echo "Superblocks: $(grep $DIR /proc/self/mountinfo | awk '{print $3}' | sort -u | wc -l)"
>>> echo "erofs_inode delta: +$(( $(grep erofs_inode /proc/slabinfo | awk '{print $2}') - INODES_BEFORE ))"
>>> echo "Slab delta: +$(( $(grep ^Slab: /proc/meminfo | awk '{print $2}') - MEM_BEFORE )) kB"
>>> ```
>>> 
>>> unpatched kernel:
>>> 
>>> Superblocks: 20
>>> erofs_inode delta: +46368
>>> Slab delta: +39952 kB
>>> 0.08user 0.82system 0:00.98elapsed 93%CPU (0avgtext+0avgdata 4796maxresident)k
>>> 19168inputs+10304outputs (24major+16672minor)pagefaults 0swaps
>>> 
>>> patched kernel:
>>> 
>>> Superblocks: 1
>>> erofs_inode delta: +2044
>>> Slab delta: +1088 kB
>>> 0.07user 0.21system 0:00.31elapsed 91%CPU (0avgtext+0avgdata 4848maxresident)k
>>> 19024inputs+6784outputs (25major+16810minor)pagefaults 0swaps
>>> 
>>> The time difference shows that sharing the superblock also benefits
>>> the page cache and inode cache, as subsequent mounts of the same image
>>> avoid re-reading the backing file.  This is particularly useful for
>>> container hosts running multiple containers from the same base image.
>>
>> Doesn't this approach break-down as soon as you get to change SB flags (through
>> e.g mount -o remount)? It will change superblock flags for all of them, right?
>>
>> (I don't think you can switch off the superblock transparently on a
>> reconfigure?)
>
> This is the same preexisting behavior for block device mounts.  That
> said, I realize this can feel confusing since EROFS is not a block
> device and allowed this so far.
>
> One way to solve this could be an explicit mount option "share_sb" that
> is opt-in and, once set, blocks any remount operations.  Would that
> work?

Would something like the following fixup on top of the previous patch be
acceptable (suggestions for better names are welcome)?

Thanks,
Giuseppe

diff --git a/fs/erofs/internal.h b/fs/erofs/internal.h
index 580f8d9f14e7..ac26daf4f78a 100644
--- a/fs/erofs/internal.h
+++ b/fs/erofs/internal.h
@@ -156,6 +156,7 @@ struct erofs_sb_info {
 #define EROFS_MOUNT_DAX_NEVER		0x00000080
 #define EROFS_MOUNT_DIRECT_IO		0x00000100
 #define EROFS_MOUNT_INODE_SHARE		0x00000200
+#define EROFS_MOUNT_SHARE_SB		0x00000400
 
 #define clear_opt(opt, option)	((opt)->mount_opt &= ~EROFS_MOUNT_##option)
 #define set_opt(opt, option)	((opt)->mount_opt |= EROFS_MOUNT_##option)
diff --git a/fs/erofs/super.c b/fs/erofs/super.c
index cbd76727f0d3..dd2a351811c8 100644
--- a/fs/erofs/super.c
+++ b/fs/erofs/super.c
@@ -386,7 +386,7 @@ static void erofs_default_options(struct erofs_sb_info *sbi)
 enum {
 	Opt_user_xattr, Opt_acl, Opt_cache_strategy, Opt_dax, Opt_dax_enum,
 	Opt_device, Opt_domain_id, Opt_directio, Opt_fsoffset, Opt_inode_share,
-	Opt_source,
+	Opt_source, Opt_share_sb,
 };
 
 static const struct constant_table erofs_param_cache_strategy[] = {
@@ -414,6 +414,7 @@ static const struct fs_parameter_spec erofs_fs_parameters[] = {
 	fsparam_flag_no("directio",		Opt_directio),
 	fsparam_u64("fsoffset",			Opt_fsoffset),
 	fsparam_flag("inode_share",		Opt_inode_share),
+	fsparam_flag("share_sb",		Opt_share_sb),
 	fsparam_file_or_string("source",	Opt_source),
 	{}
 };
@@ -560,6 +561,12 @@ static int erofs_fc_parse_param(struct fs_context *fc,
 		else
 			set_opt(&sbi->opt, INODE_SHARE);
 		break;
+	case Opt_share_sb:
+		if (!IS_ENABLED(CONFIG_EROFS_FS_BACKED_BY_FILE))
+			errorfc(fc, "%s option not supported", erofs_fs_parameters[opt].name);
+		else
+			set_opt(&sbi->opt, SHARE_SB);
+		break;
 	case Opt_source:
 		return erofs_fc_parse_source(fc, param);
 	}
@@ -798,6 +805,8 @@ static int erofs_fc_test_file_super(struct super_block *sb,
 		return 0;
 	if (!sbi->dif0.file || !new_sbi->dif0.file)
 		return 0;
+	if (!test_opt(&new_sbi->opt, SHARE_SB))
+		return 0;
 	return file_inode(sbi->dif0.file) == file_inode(new_sbi->dif0.file) &&
 	       sbi->dif0.fsoff == new_sbi->dif0.fsoff &&
 	       sbi->opt.mount_opt == new_sbi->opt.mount_opt &&
@@ -874,6 +883,9 @@ static int erofs_fc_reconfigure(struct fs_context *fc)
 
 	DBG_BUGON(!sb_rdonly(sb));
 
+	if (test_opt(&sbi->opt, SHARE_SB))
+		return -EBUSY;
+
 	if (new_sbi->domain_id)
 		erofs_info(sb, "ignoring reconfiguration for domain_id.");
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.