Re: [PATCH v2 1/4] btrfs: zoned: flush active metadata block group at btree_writepages() start

Boris Burkov <[email protected]> Thu, 23 Jul 2026 09:53:50 -0700
Newsgroups org.kernel.vger.linux-btrfs
Message-ID <[email protected]>
On Thu, Jul 23, 2026 at 05:54:25PM +0200, Johannes Thumshirn wrote:
> btree_writepages() writes the btree inode's dirty metadata in ascending
> logical address order. On a zoned filesystem only one metadata and one
> system block group is active for writing at a time, and
> check_bg_is_active() (via btrfs_check_meta_write_pointer()) pivots the
> active block group as writeback moves from one block group to the next.
> 
> If the active block group sits at a higher logical address than another
> block group that also holds dirty metadata, the ascending walk reaches
> the lower one first and, to write it, has to finish the active block
> group and activate the lower one. It cannot finish a block group that
> still has unsent IO, and during WB_SYNC_ALL && !for_sync (commit)
> writeback it deliberately refuses to wait for that IO under
> fs_info->zoned_meta_io_lock, as that can deadlock. The pivot thus cannot
> issue the submission itself either, so it gives up:
> btrfs_check_meta_write_pointer() returns -EAGAIN, which
> btrfs_write_and_wait_transaction() treats as fatal and aborts the
> transaction, forcing the filesystem read-only. This happens
> intermittently under metadata-heavy relocation (e.g. fstests btrfs/187).
> 
> Flush the active metadata and system block groups at the start of
> btree_writepages(), under the fs_info->zoned_meta_io_lock it already
> holds, so they have no unsent IO left and the later pivot can finish
> them and make forward progress.
> 
> Fixes: 13bb483d32ab ("btrfs: zoned: activate metadata block group on write time")
> Assisted-by: LLM (debugging, commit message)
> Signed-off-by: Johannes Thumshirn <[email protected]>

This version looks great, thanks.

Reviewed-by: Boris Burkov <[email protected]>
> ---
>  fs/btrfs/extent_io.c | 113 +++++++++++++++++++++++++++++++++++--------
>  1 file changed, 92 insertions(+), 21 deletions(-)
> 
> diff --git a/fs/btrfs/extent_io.c b/fs/btrfs/extent_io.c
> index d119dcf9e34b..e3125ef701ce 100644
> --- a/fs/btrfs/extent_io.c
> +++ b/fs/btrfs/extent_io.c
> @@ -2383,6 +2383,76 @@ void btrfs_btree_wait_writeback_range(struct btrfs_fs_info *fs_info, u64 start,
>  	}
>  }
>  
> +static int write_meta_extent_buffer(struct btrfs_eb_write_context *ctx,
> +				    struct writeback_control *wbc)
> +{
> +	struct extent_buffer *eb = ctx->eb;
> +	int ret;
> +
> +	ret = btrfs_check_meta_write_pointer(eb->fs_info, ctx);
> +	if (ret)
> +		return ret;
> +
> +	if (!lock_extent_buffer_for_io(eb, wbc))
> +		return 0;
> +
> +	/* Implies write in zoned mode. */
> +	if (ctx->zoned_bg) {
> +		/* Mark the last eb in the block group. */
> +		btrfs_schedule_zone_finish_bg(ctx->zoned_bg, eb);
> +		ctx->zoned_bg->meta_write_pointer += eb->len;
> +	}
> +	write_one_eb(eb, wbc);
> +	return 0;
> +}
> +
> +/*
> + * On a zoned filesystem, write out the currently dirty metadata extent buffers
> + * of @bg. Used to flush the active metadata/system block group before the
> + * ascending-address walk in btree_writepages(), so that walk can pivot the
> + * active block group away (finishing it) instead of aborting the commit; see
> + * the caller for details.
> + */
> +static void flush_active_meta_bg(struct address_space *mapping,
> +				 struct writeback_control *wbc,
> +				 struct btrfs_eb_write_context *ctx,
> +				 struct btrfs_block_group *bg)
> +{
> +	struct btrfs_fs_info *fs_info = inode_to_fs_info(mapping->host);
> +	unsigned long index = bg->start >> fs_info->nodesize_bits;
> +	unsigned long end = (btrfs_block_group_end(bg) - 1) >> fs_info->nodesize_bits;
> +	struct eb_batch batch;
> +	unsigned int nr_ebs;
> +
> +	ASSERT(btrfs_is_zoned(fs_info));
> +	lockdep_assert_held(&fs_info->zoned_meta_io_lock);
> +
> +	eb_batch_init(&batch);
> +	while (index <= end &&
> +	       (nr_ebs = buffer_tree_get_ebs_tag(fs_info, &index, end,
> +						 PAGECACHE_TAG_DIRTY, &batch))) {
> +		struct extent_buffer *eb;
> +
> +		while ((eb = eb_batch_next(&batch)) != NULL) {
> +			ctx->eb = eb;
> +
> +			/*
> +			 * If the eb is behind the write pointer (-EBUSY, e.g.
> +			 * already being written by someone else) skip it and
> +			 * carry on. Only a hole at the write pointer (-EAGAIN)
> +			 * stops the flush. The main walk in btree_writepages()
> +			 * then deals with it.
> +			 */
> +			if (write_meta_extent_buffer(ctx, wbc) == -EAGAIN) {
> +				eb_batch_release(&batch);
> +				return;
> +			}
> +		}
> +		eb_batch_release(&batch);
> +		cond_resched();
> +	}
> +}
> +
>  int btree_writepages(struct address_space *mapping, struct writeback_control *wbc)
>  {
>  	struct btrfs_eb_write_context ctx = { .wbc = wbc };
> @@ -2418,6 +2488,22 @@ int btree_writepages(struct address_space *mapping, struct writeback_control *wb
>  	else
>  		tag = PAGECACHE_TAG_DIRTY;
>  	btrfs_zoned_meta_io_lock(fs_info);
> +
> +	/*
> +	 * On a zoned filesystem, flush the currently active metadata/system
> +	 * block group(s) first, under this same lock, so the ascending-address
> +	 * walk below can pivot the active block group instead of aborting the
> +	 * transaction commit with -EAGAIN.
> +	 */
> +	if (btrfs_is_zoned(fs_info) && wbc->sync_mode == WB_SYNC_ALL &&
> +	    !wbc->for_sync) {
> +		if (fs_info->active_meta_bg)
> +			flush_active_meta_bg(mapping, wbc, &ctx,
> +					     fs_info->active_meta_bg);
> +		if (fs_info->active_system_bg)
> +			flush_active_meta_bg(mapping, wbc, &ctx,
> +					     fs_info->active_system_bg);
> +	}
>  retry:
>  	if (wbc->sync_mode == WB_SYNC_ALL)
>  		buffer_tree_tag_for_writeback(fs_info, index, end);
> @@ -2428,28 +2514,13 @@ int btree_writepages(struct address_space *mapping, struct writeback_control *wb
>  		while ((eb = eb_batch_next(&batch)) != NULL) {
>  			ctx.eb = eb;
>  
> -			ret = btrfs_check_meta_write_pointer(eb->fs_info, &ctx);
> -			if (ret) {
> -				if (ret == -EBUSY)
> -					ret = 0;
> -
> -				if (ret) {
> -					done = true;
> -					break;
> -				}
> -				continue;
> -			}
> -
> -			if (!lock_extent_buffer_for_io(eb, wbc))
> -				continue;
> -
> -			/* Implies write in zoned mode. */
> -			if (ctx.zoned_bg) {
> -				/* Mark the last eb in the block group. */
> -				btrfs_schedule_zone_finish_bg(ctx.zoned_bg, eb);
> -				ctx.zoned_bg->meta_write_pointer += eb->len;
> +			ret = write_meta_extent_buffer(&ctx, wbc);
> +			if (ret == -EBUSY) {
> +				ret = 0;
> +			} else if (ret) {
> +				done = true;
> +				break;
>  			}
> -			write_one_eb(eb, wbc);
>  		}
>  		nr_to_write_done = (wbc->nr_to_write <= 0);
>  		eb_batch_release(&batch);
> -- 
> 2.55.0
>