Re: [PATCH v5] btrfs: zoned: don't force read-only on transient -EAGAIN from reloc merge
Filipe Manana <[email protected]>
| Newsgroups | org.kernel.vger.linux-btrfs |
|---|---|
| Message-ID | <CAL3q7H6moPHk7eAVaGs4BNkSYonmnyeV8c_aqr=qkj_ki5MufQ@mail.gmail.com> |
On Tue, Jun 23, 2026 at 12:39 PM Johannes Thumshirn <[email protected]> wrote: > > On a zoned FS, btrfs_delayed_refs_rsv_refill() returns -EAGAIN whenever > the over-committed metadata plus the zone_unusable bytes exceeds the > usable size in a metadata block-group to avoid heavy over-commit of > metadata and early ENOSPC in one transaction. > > If this happens while doing reclaim, the transaction is getting > aborted. > > Treat -EAGAIN as a soft, retryable condition in case of block-group > reclaim. > > Reported-by: Damien Le Moal <[email protected]> > Fixes: 7bcb04de982f ("btrfs: zoned: cap delayed refs metadata reservation to avoid overcommit") > Signed-off-by: Johannes Thumshirn <[email protected]> Reviewed-by: Filipe Manana <[email protected]> Thanks. > --- > Changes to v4: > - Gate EAGAIN checks behind btrfs_is_zoned() > - Fix comment style > > fs/btrfs/block-group.c | 9 +++++++- > fs/btrfs/relocation.c | 51 ++++++++++++++++++++++++++++++++++++++---- > 2 files changed, 55 insertions(+), 5 deletions(-) > > diff --git a/fs/btrfs/block-group.c b/fs/btrfs/block-group.c > index ab76a5173272..9e937407d12a 100644 > --- a/fs/btrfs/block-group.c > +++ b/fs/btrfs/block-group.c > @@ -2047,6 +2047,12 @@ static int btrfs_reclaim_block_group(struct btrfs_block_group *bg, int *reclaime > > trace_btrfs_reclaim_block_group(bg); > ret = btrfs_relocate_chunk(fs_info, bg->start, false); > + if (btrfs_is_zoned(fs_info) && ret == -EAGAIN) { > + btrfs_dec_block_group_ro(bg); > + btrfs_debug(fs_info, "deferring reclaim of chunk %llu", > + bg->start); > + return ret; > + } > if (ret) { > btrfs_dec_block_group_ro(bg); > btrfs_err(fs_info, "error relocating chunk %llu", > @@ -2113,7 +2119,8 @@ void btrfs_reclaim_block_groups(struct btrfs_fs_info *fs_info, unsigned int limi > spin_unlock(&fs_info->unused_bgs_lock); > ret = btrfs_reclaim_block_group(bg, &reclaimed); > > - if (ret && !READ_ONCE(space_info->periodic_reclaim)) > + if ((btrfs_is_zoned(fs_info) && ret == -EAGAIN) || > + (ret && !READ_ONCE(space_info->periodic_reclaim))) > btrfs_link_bg_list(bg, &retry_list); > btrfs_put_block_group(bg); > > diff --git a/fs/btrfs/relocation.c b/fs/btrfs/relocation.c > index 1819d49d19c3..df69b401ca8e 100644 > --- a/fs/btrfs/relocation.c > +++ b/fs/btrfs/relocation.c > @@ -1530,6 +1530,36 @@ static void clear_reloc_root(struct btrfs_root *root) > clear_bit(BTRFS_ROOT_DEAD_RELOC_TREE, &root->state); > } > > +/* Drop the reloc trees of a relocation that is being deferred and retried. */ > +static void abort_reloc_roots(struct reloc_control *rc, struct list_head *list) > +{ > + struct btrfs_fs_info *fs_info = rc->extent_root->fs_info; > + struct btrfs_root *reloc_root, *tmp; > + > + list_for_each_entry_safe(reloc_root, tmp, list, root_list) { > + struct btrfs_root *root; > + > + root = btrfs_get_fs_root(fs_info, reloc_root->root_key.offset, > + false); > + if (!IS_ERR(root)) { > + if (root->reloc_root == reloc_root) { > + clear_reloc_root(root); > + btrfs_put_root(reloc_root); > + } > + btrfs_put_root(root); > + } > + > + btrfs_set_root_refs(&reloc_root->root_item, 0); > + memset(&reloc_root->root_item.drop_progress, 0, > + sizeof(struct btrfs_disk_key)); > + btrfs_set_root_drop_level(&reloc_root->root_item, 0); > + > + list_del_init(&reloc_root->root_list); > + list_add_tail(&reloc_root->reloc_dirty_list, > + &rc->dirty_subvol_roots); > + } > +} > + > static int clean_dirty_subvols(struct reloc_control *rc) > { > struct btrfs_root *root; > @@ -1870,7 +1900,7 @@ int prepare_to_merge(struct reloc_control *rc, int err) > } > > static noinline_for_stack > -void merge_reloc_roots(struct reloc_control *rc) > +int merge_reloc_roots(struct reloc_control *rc) > { > struct btrfs_fs_info *fs_info = rc->extent_root->fs_info; > struct btrfs_root *root; > @@ -1968,7 +1998,15 @@ void merge_reloc_roots(struct reloc_control *rc) > goto again; > } > out: > - if (ret) { > + if (btrfs_is_zoned(fs_info) && ret == -EAGAIN) { > + abort_reloc_roots(rc, &reloc_roots); > + > + /* New reloc root may be added. */ > + mutex_lock(&fs_info->reloc_mutex); > + list_splice_init(&rc->reloc_roots, &reloc_roots); > + mutex_unlock(&fs_info->reloc_mutex); > + abort_reloc_roots(rc, &reloc_roots); > + } else if (ret) { > btrfs_handle_fs_error(fs_info, ret, NULL); > free_reloc_roots(&reloc_roots); > > @@ -1994,6 +2032,7 @@ void merge_reloc_roots(struct reloc_control *rc) > * > * The remaining nodes will be cleaned up by put_reloc_control(). > */ > + return ret; > } > > static void free_block_list(struct rb_root *blocks) > @@ -3723,7 +3762,9 @@ static noinline_for_stack int relocate_block_group(struct reloc_control *rc) > */ > err = prepare_to_merge(rc, err); > > - merge_reloc_roots(rc); > + ret = merge_reloc_roots(rc); > + if (ret && !err) > + err = ret; > > rc->merge_reloc_tree = false; > unset_reloc_control(rc); > @@ -5692,7 +5733,9 @@ int btrfs_recover_relocation(struct btrfs_fs_info *fs_info) > if (ret) > goto out_unset; > > - merge_reloc_roots(rc); > + ret = merge_reloc_roots(rc); > + if (ret) > + goto out_unset; > > unset_reloc_control(rc); > > -- > 2.54.0 > >